@tanstack/ai-lovable 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +37 -0
- package/dist/esm/adapters/embedding.d.ts +16 -0
- package/dist/esm/adapters/embedding.js +66 -0
- package/dist/esm/adapters/embedding.js.map +1 -0
- package/dist/esm/adapters/factory.d.ts +16 -0
- package/dist/esm/adapters/factory.js +28 -0
- package/dist/esm/adapters/factory.js.map +1 -0
- package/dist/esm/adapters/image.d.ts +26 -0
- package/dist/esm/adapters/image.js +145 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/responses-text.d.ts +24 -0
- package/dist/esm/adapters/responses-text.js +30 -0
- package/dist/esm/adapters/responses-text.js.map +1 -0
- package/dist/esm/adapters/summarize.d.ts +9 -0
- package/dist/esm/adapters/summarize.js +17 -0
- package/dist/esm/adapters/summarize.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +20 -0
- package/dist/esm/adapters/text.js +21 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +19 -0
- package/dist/esm/adapters/transcription.js +123 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +16 -0
- package/dist/esm/adapters/tts.js +69 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +41 -0
- package/dist/esm/adapters/video.js +164 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +6 -0
- package/dist/esm/audio/tts-provider-options.d.ts +8 -0
- package/dist/esm/byok.d.ts +1 -0
- package/dist/esm/byok.js +11 -0
- package/dist/esm/byok.js.map +1 -0
- package/dist/esm/embedding/embedding-provider-options.d.ts +9 -0
- package/dist/esm/image/image-input-to-file.d.ts +8 -0
- package/dist/esm/image/image-input-to-file.js +47 -0
- package/dist/esm/image/image-input-to-file.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +12 -0
- package/dist/esm/index.d.ts +26 -0
- package/dist/esm/index.js +12 -0
- package/dist/esm/message-types.d.ts +44 -0
- package/dist/esm/model-meta.d.ts +71 -0
- package/dist/esm/model-meta.js +63 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/responses-provider-options.d.ts +3 -0
- package/dist/esm/text/text-provider-options.d.ts +17 -0
- package/dist/esm/utils/client.d.ts +14 -0
- package/dist/esm/utils/client.js +35 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +19 -0
- package/dist/esm/video/video-provider-options.js +47 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/package.json +79 -0
- package/src/adapters/embedding.ts +102 -0
- package/src/adapters/factory.ts +78 -0
- package/src/adapters/image.ts +304 -0
- package/src/adapters/responses-text.ts +66 -0
- package/src/adapters/summarize.ts +35 -0
- package/src/adapters/text.ts +49 -0
- package/src/adapters/transcription.ts +226 -0
- package/src/adapters/tts.ts +97 -0
- package/src/adapters/video.ts +303 -0
- package/src/audio/transcription-provider-options.ts +7 -0
- package/src/audio/tts-provider-options.ts +21 -0
- package/src/byok.ts +7 -0
- package/src/embedding/embedding-provider-options.ts +9 -0
- package/src/image/image-input-to-file.ts +81 -0
- package/src/image/image-provider-options.ts +17 -0
- package/src/index.ts +125 -0
- package/src/message-types.ts +45 -0
- package/src/model-meta.ts +161 -0
- package/src/text/responses-provider-options.ts +8 -0
- package/src/text/text-provider-options.ts +18 -0
- package/src/utils/client.ts +50 -0
- package/src/video/video-provider-options.ts +128 -0
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
import OpenAI from 'openai'
|
|
2
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
3
|
+
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
|
+
import { base64ToArrayBuffer, generateId } from '@tanstack/ai-utils'
|
|
5
|
+
import {
|
|
6
|
+
getLovableApiKeyFromEnv,
|
|
7
|
+
openaiRequestOptions,
|
|
8
|
+
withLovableDefaults,
|
|
9
|
+
} from '../utils/client'
|
|
10
|
+
import type {
|
|
11
|
+
TokenUsage,
|
|
12
|
+
TranscriptionOptions,
|
|
13
|
+
TranscriptionResponseFormat,
|
|
14
|
+
TranscriptionResult,
|
|
15
|
+
} from '@tanstack/ai'
|
|
16
|
+
import type OpenAI_SDK from 'openai'
|
|
17
|
+
import type { LovableTranscriptionModel } from '../model-meta'
|
|
18
|
+
import type { LovableTranscriptionProviderOptions } from '../audio/transcription-provider-options'
|
|
19
|
+
import type { LovableClientConfig } from '../utils/client'
|
|
20
|
+
|
|
21
|
+
function isPlainFormat(format: string): format is 'json' | 'text' {
|
|
22
|
+
return format === 'json' || format === 'text'
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export interface LovableTranscriptionConfig extends LovableClientConfig {}
|
|
26
|
+
|
|
27
|
+
function buildTranscriptionUsage(
|
|
28
|
+
response?: OpenAI_SDK.Audio.TranscriptionCreateResponse,
|
|
29
|
+
): TokenUsage | undefined {
|
|
30
|
+
const usage = response?.usage
|
|
31
|
+
if (!usage || usage.type === 'duration') {
|
|
32
|
+
return undefined
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const result: TokenUsage = {
|
|
36
|
+
promptTokens: usage.input_tokens || 0,
|
|
37
|
+
completionTokens: usage.output_tokens || 0,
|
|
38
|
+
totalTokens: usage.total_tokens || 0,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const inputDetails = usage.input_token_details
|
|
42
|
+
const promptTokensDetails = {
|
|
43
|
+
...(inputDetails?.audio_tokens
|
|
44
|
+
? { audioTokens: inputDetails.audio_tokens }
|
|
45
|
+
: {}),
|
|
46
|
+
...(inputDetails?.text_tokens
|
|
47
|
+
? { textTokens: inputDetails.text_tokens }
|
|
48
|
+
: {}),
|
|
49
|
+
}
|
|
50
|
+
if (Object.keys(promptTokensDetails).length > 0) {
|
|
51
|
+
result.promptTokensDetails = promptTokensDetails
|
|
52
|
+
}
|
|
53
|
+
if (usage.output_tokens) {
|
|
54
|
+
result.completionTokensDetails = { textTokens: usage.output_tokens }
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
return result
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export class LovableTranscriptionAdapter<
|
|
61
|
+
TModel extends LovableTranscriptionModel,
|
|
62
|
+
> extends BaseTranscriptionAdapter<
|
|
63
|
+
TModel,
|
|
64
|
+
LovableTranscriptionProviderOptions
|
|
65
|
+
> {
|
|
66
|
+
readonly name = 'lovable' as const
|
|
67
|
+
|
|
68
|
+
protected client: OpenAI
|
|
69
|
+
|
|
70
|
+
constructor(config: LovableTranscriptionConfig, model: TModel) {
|
|
71
|
+
super(model, {})
|
|
72
|
+
this.client = new OpenAI(withLovableDefaults(config))
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
async transcribe(
|
|
76
|
+
options: TranscriptionOptions<LovableTranscriptionProviderOptions>,
|
|
77
|
+
): Promise<TranscriptionResult> {
|
|
78
|
+
const { model, language } = options
|
|
79
|
+
|
|
80
|
+
try {
|
|
81
|
+
const request = this.buildTranscriptionRequest(options)
|
|
82
|
+
|
|
83
|
+
options.logger.request(
|
|
84
|
+
`activity=transcription provider=${this.name} model=${model}`,
|
|
85
|
+
{ provider: this.name, model },
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
const response = await this.client.audio.transcriptions.create(
|
|
89
|
+
request,
|
|
90
|
+
openaiRequestOptions(options.abortSignal),
|
|
91
|
+
)
|
|
92
|
+
const usage =
|
|
93
|
+
typeof response === 'string'
|
|
94
|
+
? undefined
|
|
95
|
+
: buildTranscriptionUsage(response)
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
id: generateId(this.name),
|
|
99
|
+
model,
|
|
100
|
+
text: typeof response === 'string' ? response : response.text,
|
|
101
|
+
...(language !== undefined && { language }),
|
|
102
|
+
...(usage !== undefined && { usage }),
|
|
103
|
+
}
|
|
104
|
+
} catch (error: unknown) {
|
|
105
|
+
options.logger.errors(`${this.name}.transcribe fatal`, {
|
|
106
|
+
error: toRunErrorPayload(error, `${this.name}.transcribe failed`),
|
|
107
|
+
source: `${this.name}.transcribe`,
|
|
108
|
+
})
|
|
109
|
+
throw error
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
private buildTranscriptionRequest(
|
|
114
|
+
options: TranscriptionOptions<LovableTranscriptionProviderOptions>,
|
|
115
|
+
): OpenAI_SDK.Audio.TranscriptionCreateParamsNonStreaming {
|
|
116
|
+
const { model, audio, language, prompt, responseFormat, modelOptions } =
|
|
117
|
+
options
|
|
118
|
+
const file = this.prepareAudioFile(audio)
|
|
119
|
+
const topLevelResponseFormat = responseFormat
|
|
120
|
+
const effectiveResponseFormat =
|
|
121
|
+
topLevelResponseFormat ?? modelOptions?.response_format
|
|
122
|
+
|
|
123
|
+
if (
|
|
124
|
+
topLevelResponseFormat !== undefined &&
|
|
125
|
+
modelOptions?.response_format !== undefined &&
|
|
126
|
+
topLevelResponseFormat !== modelOptions.response_format
|
|
127
|
+
) {
|
|
128
|
+
throw new Error(
|
|
129
|
+
`Conflicting response formats: responseFormat="${topLevelResponseFormat}" and modelOptions.response_format="${modelOptions.response_format}". Provide only one.`,
|
|
130
|
+
)
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (
|
|
134
|
+
effectiveResponseFormat !== undefined &&
|
|
135
|
+
!isPlainFormat(effectiveResponseFormat)
|
|
136
|
+
) {
|
|
137
|
+
throw new Error(
|
|
138
|
+
`lovable: model "${model}" only supports json and text response formats; received "${effectiveResponseFormat}".`,
|
|
139
|
+
)
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
const request: OpenAI_SDK.Audio.TranscriptionCreateParamsNonStreaming = {
|
|
143
|
+
...modelOptions,
|
|
144
|
+
model,
|
|
145
|
+
file,
|
|
146
|
+
}
|
|
147
|
+
delete request.stream
|
|
148
|
+
if (language !== undefined) {
|
|
149
|
+
request.language = language
|
|
150
|
+
}
|
|
151
|
+
if (prompt !== undefined) {
|
|
152
|
+
request.prompt = prompt
|
|
153
|
+
}
|
|
154
|
+
request.response_format = mapResponseFormat(effectiveResponseFormat)
|
|
155
|
+
|
|
156
|
+
return request
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
protected prepareAudioFile(audio: string | File | Blob | ArrayBuffer): File {
|
|
160
|
+
if (typeof File !== 'undefined' && audio instanceof File) {
|
|
161
|
+
return audio
|
|
162
|
+
}
|
|
163
|
+
if (typeof Blob !== 'undefined' && audio instanceof Blob) {
|
|
164
|
+
this.ensureFileSupport()
|
|
165
|
+
return new File([audio], 'audio.mp3', {
|
|
166
|
+
type: audio.type || 'audio/mpeg',
|
|
167
|
+
})
|
|
168
|
+
}
|
|
169
|
+
if (typeof ArrayBuffer !== 'undefined' && audio instanceof ArrayBuffer) {
|
|
170
|
+
this.ensureFileSupport()
|
|
171
|
+
return new File([audio], 'audio.mp3', { type: 'audio/mpeg' })
|
|
172
|
+
}
|
|
173
|
+
if (typeof audio === 'string') {
|
|
174
|
+
this.ensureFileSupport()
|
|
175
|
+
|
|
176
|
+
if (audio.startsWith('data:')) {
|
|
177
|
+
const parts = audio.split(',')
|
|
178
|
+
const header = parts[0]
|
|
179
|
+
const base64Data = parts[1] || ''
|
|
180
|
+
const mimeMatch = header?.match(/data:([^;]+)/)
|
|
181
|
+
const mimeType = mimeMatch?.[1] || 'audio/mpeg'
|
|
182
|
+
const bytes = base64ToArrayBuffer(base64Data)
|
|
183
|
+
const extension = mimeType.split('/')[1] || 'mp3'
|
|
184
|
+
return new File([bytes], `audio.${extension}`, { type: mimeType })
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const bytes = base64ToArrayBuffer(audio)
|
|
188
|
+
return new File([bytes], 'audio.mp3', { type: 'audio/mpeg' })
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
throw new Error('Invalid audio input type')
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
private ensureFileSupport(): void {
|
|
195
|
+
if (typeof File === 'undefined') {
|
|
196
|
+
throw new Error(
|
|
197
|
+
'`File` is not available in this environment. ' +
|
|
198
|
+
'Use Node.js 20 or newer, or pass a File object directly.',
|
|
199
|
+
)
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function mapResponseFormat(
|
|
205
|
+
format?: TranscriptionResponseFormat,
|
|
206
|
+
): 'json' | 'text' {
|
|
207
|
+
if (format === 'text') return 'text'
|
|
208
|
+
return 'json'
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
export function createLovableTranscription<
|
|
212
|
+
TModel extends LovableTranscriptionModel,
|
|
213
|
+
>(
|
|
214
|
+
model: TModel,
|
|
215
|
+
apiKey: string,
|
|
216
|
+
config?: Omit<LovableTranscriptionConfig, 'apiKey'>,
|
|
217
|
+
): LovableTranscriptionAdapter<TModel> {
|
|
218
|
+
return new LovableTranscriptionAdapter({ apiKey, ...config }, model)
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
export function lovableTranscription<TModel extends LovableTranscriptionModel>(
|
|
222
|
+
model: TModel,
|
|
223
|
+
config?: Omit<LovableTranscriptionConfig, 'apiKey'>,
|
|
224
|
+
): LovableTranscriptionAdapter<TModel> {
|
|
225
|
+
return createLovableTranscription(model, getLovableApiKeyFromEnv(), config)
|
|
226
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
import OpenAI from 'openai'
|
|
2
|
+
import { BaseTTSAdapter } from '@tanstack/ai/adapters'
|
|
3
|
+
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
|
+
import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
|
|
5
|
+
import {
|
|
6
|
+
getLovableApiKeyFromEnv,
|
|
7
|
+
openaiRequestOptions,
|
|
8
|
+
withLovableDefaults,
|
|
9
|
+
} from '../utils/client'
|
|
10
|
+
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
11
|
+
import type OpenAI_SDK from 'openai'
|
|
12
|
+
import type { LovableTTSModel } from '../model-meta'
|
|
13
|
+
import type { LovableTTSProviderOptions } from '../audio/tts-provider-options'
|
|
14
|
+
import type { LovableClientConfig } from '../utils/client'
|
|
15
|
+
|
|
16
|
+
export interface LovableTTSConfig extends LovableClientConfig {}
|
|
17
|
+
|
|
18
|
+
const CONTENT_TYPES: Record<string, string> = {
|
|
19
|
+
mp3: 'audio/mpeg',
|
|
20
|
+
opus: 'audio/opus',
|
|
21
|
+
aac: 'audio/aac',
|
|
22
|
+
flac: 'audio/flac',
|
|
23
|
+
wav: 'audio/wav',
|
|
24
|
+
pcm: 'audio/pcm',
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export class LovableTTSAdapter<
|
|
28
|
+
TModel extends LovableTTSModel,
|
|
29
|
+
> extends BaseTTSAdapter<TModel, LovableTTSProviderOptions> {
|
|
30
|
+
readonly name = 'lovable' as const
|
|
31
|
+
|
|
32
|
+
protected client: OpenAI
|
|
33
|
+
|
|
34
|
+
constructor(config: LovableTTSConfig, model: TModel) {
|
|
35
|
+
super(model, {})
|
|
36
|
+
this.client = new OpenAI(withLovableDefaults(config))
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
async generateSpeech(
|
|
40
|
+
options: TTSOptions<LovableTTSProviderOptions>,
|
|
41
|
+
): Promise<TTSResult> {
|
|
42
|
+
const { model, text, voice, format, speed, modelOptions } = options
|
|
43
|
+
|
|
44
|
+
const request: OpenAI_SDK.Audio.SpeechCreateParams = {
|
|
45
|
+
model,
|
|
46
|
+
input: text,
|
|
47
|
+
voice: voice || 'alloy',
|
|
48
|
+
response_format: format,
|
|
49
|
+
...(speed !== undefined && { speed }),
|
|
50
|
+
...(modelOptions ?? {}),
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
try {
|
|
54
|
+
options.logger.request(
|
|
55
|
+
`activity=tts provider=${this.name} model=${model} format=${request.response_format ?? 'default'} voice=${request.voice}`,
|
|
56
|
+
{ provider: this.name, model },
|
|
57
|
+
)
|
|
58
|
+
const response = await this.client.audio.speech.create(
|
|
59
|
+
request,
|
|
60
|
+
openaiRequestOptions(options.abortSignal),
|
|
61
|
+
)
|
|
62
|
+
const arrayBuffer = await response.arrayBuffer()
|
|
63
|
+
const base64 = arrayBufferToBase64(arrayBuffer)
|
|
64
|
+
const outputFormat = (request.response_format as string) || 'mp3'
|
|
65
|
+
const contentType = CONTENT_TYPES[outputFormat] || 'audio/mpeg'
|
|
66
|
+
|
|
67
|
+
return {
|
|
68
|
+
id: generateId(this.name),
|
|
69
|
+
model,
|
|
70
|
+
audio: base64,
|
|
71
|
+
format: outputFormat,
|
|
72
|
+
contentType,
|
|
73
|
+
}
|
|
74
|
+
} catch (error: unknown) {
|
|
75
|
+
options.logger.errors(`${this.name}.generateSpeech fatal`, {
|
|
76
|
+
error: toRunErrorPayload(error, `${this.name}.generateSpeech failed`),
|
|
77
|
+
source: `${this.name}.generateSpeech`,
|
|
78
|
+
})
|
|
79
|
+
throw error
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function createLovableSpeech<TModel extends LovableTTSModel>(
|
|
85
|
+
model: TModel,
|
|
86
|
+
apiKey: string,
|
|
87
|
+
config?: Omit<LovableTTSConfig, 'apiKey'>,
|
|
88
|
+
): LovableTTSAdapter<TModel> {
|
|
89
|
+
return new LovableTTSAdapter({ apiKey, ...config }, model)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function lovableSpeech<TModel extends LovableTTSModel>(
|
|
93
|
+
model: TModel,
|
|
94
|
+
config?: Omit<LovableTTSConfig, 'apiKey'>,
|
|
95
|
+
): LovableTTSAdapter<TModel> {
|
|
96
|
+
return createLovableSpeech(model, getLovableApiKeyFromEnv(), config)
|
|
97
|
+
}
|
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
import OpenAI from 'openai'
|
|
2
|
+
import { resolveMediaPrompt } from '@tanstack/ai'
|
|
3
|
+
import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
|
|
4
|
+
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
5
|
+
import { arrayBufferToBase64 } from '@tanstack/ai-utils'
|
|
6
|
+
import {
|
|
7
|
+
getLovableApiKeyFromEnv,
|
|
8
|
+
openaiRequestOptions,
|
|
9
|
+
withLovableDefaults,
|
|
10
|
+
} from '../utils/client'
|
|
11
|
+
import { imagePartToFile } from '../image/image-input-to-file'
|
|
12
|
+
import {
|
|
13
|
+
toApiSeconds,
|
|
14
|
+
validateHighResDuration,
|
|
15
|
+
validateVideoSeconds,
|
|
16
|
+
validateVideoSize,
|
|
17
|
+
} from '../video/video-provider-options'
|
|
18
|
+
import type { DurationOptions } from '@tanstack/ai/adapters'
|
|
19
|
+
import type {
|
|
20
|
+
VideoGenerationOptions,
|
|
21
|
+
VideoJobResult,
|
|
22
|
+
VideoStatusResult,
|
|
23
|
+
VideoUrlResult,
|
|
24
|
+
} from '@tanstack/ai'
|
|
25
|
+
import type OpenAI_SDK from 'openai'
|
|
26
|
+
import type {
|
|
27
|
+
LovableVideoModel,
|
|
28
|
+
LovableVideoModelDurationByName,
|
|
29
|
+
LovableVideoModelInputModalitiesByName,
|
|
30
|
+
LovableVideoModelProviderOptionsByName,
|
|
31
|
+
LovableVideoModelSizeByName,
|
|
32
|
+
} from '../model-meta'
|
|
33
|
+
import type {
|
|
34
|
+
LovableVideoDuration,
|
|
35
|
+
LovableVideoProviderOptions,
|
|
36
|
+
} from '../video/video-provider-options'
|
|
37
|
+
import type { LovableClientConfig } from '../utils/client'
|
|
38
|
+
|
|
39
|
+
const LARGE_MEDIA_BUFFER_BYTES = 10 * 1024 * 1024
|
|
40
|
+
const VIDEO_DURATIONS = [
|
|
41
|
+
4, 6, 8,
|
|
42
|
+
] as const satisfies ReadonlyArray<LovableVideoDuration>
|
|
43
|
+
|
|
44
|
+
function warnIfLargeMediaBuffer(byteLength: number, source: string): void {
|
|
45
|
+
if (byteLength <= LARGE_MEDIA_BUFFER_BYTES) return
|
|
46
|
+
console.warn(
|
|
47
|
+
`[lovable.${source}] downloaded ${(byteLength / 1024 / 1024).toFixed(1)} MiB into memory before base64 encoding. ` +
|
|
48
|
+
`Workers/serverless runtimes commonly run out of memory above ~10 MiB. ` +
|
|
49
|
+
`Consider streaming the video through a CDN or your own storage layer instead.`,
|
|
50
|
+
)
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
55
|
+
*/
|
|
56
|
+
export interface LovableVideoConfig extends LovableClientConfig {
|
|
57
|
+
/**
|
|
58
|
+
* Opt into fetching HTTP(S) image URL inputs for `input_reference`.
|
|
59
|
+
* The endpoint requires uploaded file bytes, so an HTTP(S) URL has to be
|
|
60
|
+
* downloaded and buffered in memory. When `false` (the default), HTTP(S)
|
|
61
|
+
* URL image inputs throw.
|
|
62
|
+
*/
|
|
63
|
+
allowUrlFetch?: boolean
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
68
|
+
*/
|
|
69
|
+
export class LovableVideoAdapter<
|
|
70
|
+
TModel extends LovableVideoModel,
|
|
71
|
+
> extends BaseVideoAdapter<
|
|
72
|
+
TModel,
|
|
73
|
+
LovableVideoProviderOptions,
|
|
74
|
+
LovableVideoModelProviderOptionsByName,
|
|
75
|
+
LovableVideoModelSizeByName,
|
|
76
|
+
LovableVideoModelInputModalitiesByName,
|
|
77
|
+
LovableVideoModelDurationByName
|
|
78
|
+
> {
|
|
79
|
+
readonly name = 'lovable' as const
|
|
80
|
+
|
|
81
|
+
protected client: OpenAI
|
|
82
|
+
protected clientConfig: LovableVideoConfig
|
|
83
|
+
|
|
84
|
+
constructor(config: LovableVideoConfig, model: TModel) {
|
|
85
|
+
super({}, model)
|
|
86
|
+
this.clientConfig = config
|
|
87
|
+
const { allowUrlFetch: _allowUrlFetch, ...clientOptions } = config
|
|
88
|
+
this.client = new OpenAI(withLovableDefaults(clientOptions))
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
async createVideoJob(
|
|
92
|
+
options: VideoGenerationOptions<
|
|
93
|
+
LovableVideoProviderOptions,
|
|
94
|
+
LovableVideoModelSizeByName[TModel],
|
|
95
|
+
LovableVideoModelDurationByName[TModel]
|
|
96
|
+
>,
|
|
97
|
+
): Promise<VideoJobResult> {
|
|
98
|
+
const { model, size, duration, modelOptions } = options
|
|
99
|
+
|
|
100
|
+
const resolvedSize = size ?? modelOptions?.size
|
|
101
|
+
validateVideoSize(model, resolvedSize)
|
|
102
|
+
const seconds = duration ?? modelOptions?.seconds
|
|
103
|
+
validateVideoSeconds(model, seconds)
|
|
104
|
+
validateHighResDuration(model, resolvedSize, seconds)
|
|
105
|
+
|
|
106
|
+
const resolved = resolveMediaPrompt(options.prompt)
|
|
107
|
+
|
|
108
|
+
if (resolved.videos.length > 0) {
|
|
109
|
+
throw new Error(
|
|
110
|
+
`${this.name}.createVideoJob does not support video prompt parts (model: ${model}).`,
|
|
111
|
+
)
|
|
112
|
+
}
|
|
113
|
+
if (resolved.audios.length > 0) {
|
|
114
|
+
throw new Error(
|
|
115
|
+
`${this.name}.createVideoJob does not support audio prompt parts (model: ${model}).`,
|
|
116
|
+
)
|
|
117
|
+
}
|
|
118
|
+
if (resolved.images.length > 1) {
|
|
119
|
+
throw new Error(
|
|
120
|
+
`${this.name}: video models accept at most one input_reference image; received ${resolved.images.length}.`,
|
|
121
|
+
)
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const request: OpenAI_SDK.Videos.VideoCreateParams = {
|
|
125
|
+
model,
|
|
126
|
+
prompt: resolved.text,
|
|
127
|
+
}
|
|
128
|
+
const [inputReference] = resolved.images
|
|
129
|
+
if (inputReference) {
|
|
130
|
+
request.input_reference = await imagePartToFile(
|
|
131
|
+
inputReference,
|
|
132
|
+
'input-reference',
|
|
133
|
+
this.clientConfig.allowUrlFetch ?? false,
|
|
134
|
+
options.abortSignal,
|
|
135
|
+
)
|
|
136
|
+
}
|
|
137
|
+
if (resolvedSize) {
|
|
138
|
+
// Gateway Veo sizes include 1080p and 4K, which are not in the OpenAI SDK union.
|
|
139
|
+
request.size = resolvedSize as OpenAI_SDK.Videos.VideoSize
|
|
140
|
+
}
|
|
141
|
+
if (seconds !== undefined) {
|
|
142
|
+
const apiSeconds = toApiSeconds(seconds)
|
|
143
|
+
if (apiSeconds !== undefined) {
|
|
144
|
+
// Gateway Veo clips can be 6 seconds. The OpenAI SDK union is 4/8/12.
|
|
145
|
+
request.seconds = apiSeconds as OpenAI_SDK.Videos.VideoSeconds
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
try {
|
|
150
|
+
options.logger.request(
|
|
151
|
+
`activity=video.create provider=${this.name} model=${model} size=${request.size ?? 'default'} seconds=${request.seconds ?? 'default'}`,
|
|
152
|
+
{ provider: this.name, model },
|
|
153
|
+
)
|
|
154
|
+
const response = await this.client.videos.create(
|
|
155
|
+
request,
|
|
156
|
+
openaiRequestOptions(options.abortSignal),
|
|
157
|
+
)
|
|
158
|
+
return { jobId: response.id, model }
|
|
159
|
+
} catch (error: unknown) {
|
|
160
|
+
options.logger.errors(`${this.name}.createVideoJob fatal`, {
|
|
161
|
+
error: toRunErrorPayload(error, `${this.name}.createVideoJob failed`),
|
|
162
|
+
source: `${this.name}.createVideoJob`,
|
|
163
|
+
})
|
|
164
|
+
throw error
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
async getVideoStatus(jobId: string): Promise<VideoStatusResult> {
|
|
169
|
+
try {
|
|
170
|
+
const response = await this.client.videos.retrieve(jobId)
|
|
171
|
+
return {
|
|
172
|
+
jobId,
|
|
173
|
+
status: this.mapStatus(response.status),
|
|
174
|
+
progress: response.progress,
|
|
175
|
+
...(response.error?.message !== undefined && {
|
|
176
|
+
error: response.error.message,
|
|
177
|
+
}),
|
|
178
|
+
}
|
|
179
|
+
} catch (error: unknown) {
|
|
180
|
+
if (isHttpStatus(error, 404)) {
|
|
181
|
+
return { jobId, status: 'failed', error: 'Job not found' }
|
|
182
|
+
}
|
|
183
|
+
throw error
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
async getVideoUrl(jobId: string): Promise<VideoUrlResult> {
|
|
188
|
+
try {
|
|
189
|
+
const videoInfo = await this.client.videos.retrieve(jobId)
|
|
190
|
+
const directUrl = videoResourceUrl(videoInfo)
|
|
191
|
+
if (directUrl) {
|
|
192
|
+
return {
|
|
193
|
+
jobId,
|
|
194
|
+
url: directUrl,
|
|
195
|
+
...(videoInfo.expires_at != null && {
|
|
196
|
+
expiresAt: new Date(videoInfo.expires_at * 1000),
|
|
197
|
+
}),
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
const contentResponse = await this.client.videos.downloadContent(jobId)
|
|
202
|
+
return dataUrlFromResponse(
|
|
203
|
+
jobId,
|
|
204
|
+
contentResponse,
|
|
205
|
+
'video.downloadContent',
|
|
206
|
+
)
|
|
207
|
+
} catch (error: unknown) {
|
|
208
|
+
if (isHttpStatus(error, 404)) {
|
|
209
|
+
throw new Error(`Video job not found: ${jobId}`)
|
|
210
|
+
}
|
|
211
|
+
if (isHttpStatus(error, 400)) {
|
|
212
|
+
throw new Error(
|
|
213
|
+
`Video is not ready for download. Check status first. Job ID: ${jobId}`,
|
|
214
|
+
)
|
|
215
|
+
}
|
|
216
|
+
throw error
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
override availableDurations(): DurationOptions<LovableVideoDuration> {
|
|
221
|
+
return { kind: 'discrete', values: VIDEO_DURATIONS }
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
override snapDuration(seconds: number): LovableVideoDuration | undefined {
|
|
225
|
+
return snapToDurationOption(seconds, this.availableDurations())
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
protected mapStatus(
|
|
229
|
+
apiStatus: string,
|
|
230
|
+
): 'pending' | 'processing' | 'completed' | 'failed' {
|
|
231
|
+
switch (apiStatus) {
|
|
232
|
+
case 'queued':
|
|
233
|
+
case 'pending':
|
|
234
|
+
return 'pending'
|
|
235
|
+
case 'processing':
|
|
236
|
+
case 'in_progress':
|
|
237
|
+
return 'processing'
|
|
238
|
+
case 'completed':
|
|
239
|
+
case 'succeeded':
|
|
240
|
+
return 'completed'
|
|
241
|
+
case 'failed':
|
|
242
|
+
case 'error':
|
|
243
|
+
case 'cancelled':
|
|
244
|
+
return 'failed'
|
|
245
|
+
default:
|
|
246
|
+
return 'processing'
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
async function dataUrlFromResponse(
|
|
252
|
+
jobId: string,
|
|
253
|
+
contentResponse: Response,
|
|
254
|
+
source: string,
|
|
255
|
+
): Promise<VideoUrlResult> {
|
|
256
|
+
const videoBlob = await contentResponse.blob()
|
|
257
|
+
const buffer = await videoBlob.arrayBuffer()
|
|
258
|
+
warnIfLargeMediaBuffer(buffer.byteLength, source)
|
|
259
|
+
const base64 = arrayBufferToBase64(buffer)
|
|
260
|
+
const mimeType = contentResponse.headers.get('content-type') || 'video/mp4'
|
|
261
|
+
return {
|
|
262
|
+
jobId,
|
|
263
|
+
url: `data:${mimeType};base64,${base64}`,
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
function videoResourceUrl(video: OpenAI_SDK.Videos.Video): string | undefined {
|
|
268
|
+
const extra = video as OpenAI_SDK.Videos.Video & { url?: string }
|
|
269
|
+
if (extra.url && extra.url.length > 0) {
|
|
270
|
+
return extra.url
|
|
271
|
+
}
|
|
272
|
+
return undefined
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
function isHttpStatus(error: unknown, status: number): boolean {
|
|
276
|
+
return (
|
|
277
|
+
typeof error === 'object' &&
|
|
278
|
+
error !== null &&
|
|
279
|
+
'status' in error &&
|
|
280
|
+
(error as { status: unknown }).status === status
|
|
281
|
+
)
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
286
|
+
*/
|
|
287
|
+
export function createLovableVideo<TModel extends LovableVideoModel>(
|
|
288
|
+
model: TModel,
|
|
289
|
+
apiKey: string,
|
|
290
|
+
config?: Omit<LovableVideoConfig, 'apiKey'>,
|
|
291
|
+
): LovableVideoAdapter<TModel> {
|
|
292
|
+
return new LovableVideoAdapter({ apiKey, ...config }, model)
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
297
|
+
*/
|
|
298
|
+
export function lovableVideo<TModel extends LovableVideoModel>(
|
|
299
|
+
model: TModel,
|
|
300
|
+
config?: Omit<LovableVideoConfig, 'apiKey'>,
|
|
301
|
+
): LovableVideoAdapter<TModel> {
|
|
302
|
+
return createLovableVideo(model, getLovableApiKeyFromEnv(), config)
|
|
303
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
export type LovableTTSVoice =
|
|
2
|
+
| 'alloy'
|
|
3
|
+
| 'ash'
|
|
4
|
+
| 'ballad'
|
|
5
|
+
| 'coral'
|
|
6
|
+
| 'echo'
|
|
7
|
+
| 'fable'
|
|
8
|
+
| 'onyx'
|
|
9
|
+
| 'nova'
|
|
10
|
+
| 'sage'
|
|
11
|
+
| 'shimmer'
|
|
12
|
+
| 'verse'
|
|
13
|
+
|
|
14
|
+
export type LovableTTSFormat = 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'
|
|
15
|
+
|
|
16
|
+
export interface LovableTTSProviderOptions {
|
|
17
|
+
/**
|
|
18
|
+
* Extra voice direction in plain language, for example "speak slowly and warmly".
|
|
19
|
+
*/
|
|
20
|
+
instructions?: string
|
|
21
|
+
}
|
package/src/byok.ts
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider options for Lovable embedding models.
|
|
3
|
+
*
|
|
4
|
+
* `dimensions` is a top-level option on `embed()`. `encoding_format` is
|
|
5
|
+
* pinned to `float` so vectors are always `number[]`.
|
|
6
|
+
*/
|
|
7
|
+
export interface LovableEmbeddingProviderOptions {
|
|
8
|
+
user?: string
|
|
9
|
+
}
|