@tanstack/ai-fal 0.6.17 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +32 -0
- package/dist/esm/adapters/audio.js +80 -0
- package/dist/esm/adapters/audio.js.map +1 -0
- package/dist/esm/adapters/image.js +18 -7
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/speech.d.ts +32 -0
- package/dist/esm/adapters/speech.js +84 -0
- package/dist/esm/adapters/speech.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +32 -0
- package/dist/esm/adapters/transcription.js +90 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/image/image-provider-options.js +10 -6
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +4 -1
- package/dist/esm/index.js +9 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +15 -0
- package/dist/esm/utils/client.d.ts +31 -0
- package/dist/esm/utils/client.js +79 -13
- package/dist/esm/utils/client.js.map +1 -1
- package/dist/esm/utils/index.d.ts +1 -1
- package/package.json +9 -4
- package/src/adapters/audio.ts +148 -0
- package/src/adapters/image.ts +29 -9
- package/src/adapters/speech.ts +147 -0
- package/src/adapters/transcription.ts +169 -0
- package/src/image/image-provider-options.ts +10 -6
- package/src/index.ts +24 -0
- package/src/model-meta.ts +27 -0
- package/src/utils/client.ts +121 -13
- package/src/utils/index.ts +4 -0
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
import { fal } from '@fal-ai/client'
|
|
2
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
3
|
+
import {
|
|
4
|
+
configureFalClient,
|
|
5
|
+
dataUrlToBlob,
|
|
6
|
+
generateId as utilGenerateId,
|
|
7
|
+
} from '../utils'
|
|
8
|
+
import type { OutputType, Result } from '@fal-ai/client'
|
|
9
|
+
import type {
|
|
10
|
+
TranscriptionOptions,
|
|
11
|
+
TranscriptionResult,
|
|
12
|
+
TranscriptionSegment,
|
|
13
|
+
} from '@tanstack/ai'
|
|
14
|
+
import type { FalClientConfig } from '../utils'
|
|
15
|
+
import type { FalModel, FalModelInput } from '../model-meta'
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Provider options for transcription, excluding fields TanStack AI handles.
|
|
19
|
+
*/
|
|
20
|
+
export type FalTranscriptionProviderOptions<TModel extends string> = Omit<
|
|
21
|
+
FalModelInput<TModel>,
|
|
22
|
+
'audio_url'
|
|
23
|
+
>
|
|
24
|
+
|
|
25
|
+
interface FalChunk {
|
|
26
|
+
text: string
|
|
27
|
+
timestamp?: [number, number] | null
|
|
28
|
+
speaker?: string
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* fal.ai transcription (speech-to-text) adapter.
|
|
33
|
+
*
|
|
34
|
+
* Supports fal.ai STT models like whisper, wizper, etc.
|
|
35
|
+
*
|
|
36
|
+
* @example
|
|
37
|
+
* ```typescript
|
|
38
|
+
* const adapter = falTranscription('fal-ai/whisper')
|
|
39
|
+
* const result = await generateTranscription({
|
|
40
|
+
* adapter,
|
|
41
|
+
* audio: 'https://example.com/audio.mp3',
|
|
42
|
+
* language: 'en',
|
|
43
|
+
* })
|
|
44
|
+
* ```
|
|
45
|
+
*/
|
|
46
|
+
export class FalTranscriptionAdapter<
|
|
47
|
+
TModel extends FalModel,
|
|
48
|
+
> extends BaseTranscriptionAdapter<
|
|
49
|
+
TModel,
|
|
50
|
+
FalTranscriptionProviderOptions<TModel>
|
|
51
|
+
> {
|
|
52
|
+
readonly name = 'fal' as const
|
|
53
|
+
|
|
54
|
+
constructor(model: TModel, config?: FalClientConfig) {
|
|
55
|
+
super(model, {})
|
|
56
|
+
configureFalClient(config)
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
async transcribe(
|
|
60
|
+
options: TranscriptionOptions<FalTranscriptionProviderOptions<TModel>>,
|
|
61
|
+
): Promise<TranscriptionResult> {
|
|
62
|
+
const { logger } = options
|
|
63
|
+
logger.request(
|
|
64
|
+
`activity=generateTranscription provider=fal model=${this.model}`,
|
|
65
|
+
{
|
|
66
|
+
provider: 'fal',
|
|
67
|
+
model: this.model,
|
|
68
|
+
},
|
|
69
|
+
)
|
|
70
|
+
try {
|
|
71
|
+
const input = this.buildInput(options)
|
|
72
|
+
const result = await fal.subscribe(this.model, { input })
|
|
73
|
+
return this.transformResponse(result)
|
|
74
|
+
} catch (error) {
|
|
75
|
+
logger.errors('fal.generateTranscription fatal', {
|
|
76
|
+
error,
|
|
77
|
+
source: 'fal.generateTranscription',
|
|
78
|
+
})
|
|
79
|
+
throw error
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
private buildInput(
|
|
84
|
+
options: TranscriptionOptions<FalTranscriptionProviderOptions<TModel>>,
|
|
85
|
+
): FalModelInput<TModel> {
|
|
86
|
+
// fal-client auto-uploads Blob/File inputs via fal.storage.upload, but
|
|
87
|
+
// passes strings through unchanged — so a `data:` URL would reach fal's
|
|
88
|
+
// API and get rejected with a 422 "Unsupported data URL". Decode data
|
|
89
|
+
// URLs to a Blob up front so the auto-upload path handles them.
|
|
90
|
+
let audioInput: string | Blob | File
|
|
91
|
+
if (options.audio instanceof ArrayBuffer) {
|
|
92
|
+
audioInput = new Blob([options.audio])
|
|
93
|
+
} else if (typeof options.audio === 'string') {
|
|
94
|
+
audioInput = dataUrlToBlob(options.audio) ?? options.audio
|
|
95
|
+
} else {
|
|
96
|
+
audioInput = options.audio
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
const input = {
|
|
100
|
+
...options.modelOptions,
|
|
101
|
+
audio_url: audioInput,
|
|
102
|
+
...(options.language ? { language: options.language } : {}),
|
|
103
|
+
...(options.prompt ? { prompt: options.prompt } : {}),
|
|
104
|
+
} as FalModelInput<TModel>
|
|
105
|
+
return input
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
protected override generateId(): string {
|
|
109
|
+
return utilGenerateId(this.name)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
private transformResponse(
|
|
113
|
+
response: Result<OutputType<TModel>>,
|
|
114
|
+
): TranscriptionResult {
|
|
115
|
+
const data = response.data as Record<string, unknown>
|
|
116
|
+
|
|
117
|
+
const text = (data.text as string) || ''
|
|
118
|
+
|
|
119
|
+
// Map fal chunks to TanStack segments. fal whisper can return
|
|
120
|
+
// `timestamp: null` on some chunks (e.g. when word-level timing is
|
|
121
|
+
// disabled), and the runtime payload is not actually constrained to a
|
|
122
|
+
// 2-tuple — treat it as unknown and validate before indexing.
|
|
123
|
+
let segments: Array<TranscriptionSegment> | undefined
|
|
124
|
+
const chunks = data.chunks as Array<FalChunk> | undefined
|
|
125
|
+
if (chunks && Array.isArray(chunks)) {
|
|
126
|
+
segments = chunks.flatMap((chunk, index) => {
|
|
127
|
+
const ts = chunk.timestamp as unknown
|
|
128
|
+
if (
|
|
129
|
+
!Array.isArray(ts) ||
|
|
130
|
+
ts.length < 2 ||
|
|
131
|
+
typeof ts[0] !== 'number' ||
|
|
132
|
+
typeof ts[1] !== 'number'
|
|
133
|
+
) {
|
|
134
|
+
return []
|
|
135
|
+
}
|
|
136
|
+
return [
|
|
137
|
+
{
|
|
138
|
+
id: index,
|
|
139
|
+
start: ts[0],
|
|
140
|
+
end: ts[1],
|
|
141
|
+
text: chunk.text,
|
|
142
|
+
...(chunk.speaker ? { speaker: chunk.speaker } : {}),
|
|
143
|
+
},
|
|
144
|
+
]
|
|
145
|
+
})
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// Extract language from response
|
|
149
|
+
const language =
|
|
150
|
+
(data.language as string) ||
|
|
151
|
+
(data.inferred_languages as Array<string> | undefined)?.[0] ||
|
|
152
|
+
(data.languages as Array<string> | undefined)?.[0]
|
|
153
|
+
|
|
154
|
+
return {
|
|
155
|
+
id: response.requestId || this.generateId(),
|
|
156
|
+
model: this.model,
|
|
157
|
+
text,
|
|
158
|
+
language,
|
|
159
|
+
segments,
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
export function falTranscription<TModel extends FalModel>(
|
|
165
|
+
model: TModel,
|
|
166
|
+
config?: FalClientConfig,
|
|
167
|
+
): FalTranscriptionAdapter<TModel> {
|
|
168
|
+
return new FalTranscriptionAdapter(model, config)
|
|
169
|
+
}
|
|
@@ -8,13 +8,17 @@ export function mapSizeToFalFormat<TModel extends string>(
|
|
|
8
8
|
// "16:9_4K" → { aspect_ratio, resolution }
|
|
9
9
|
// "16:9" → { aspect_ratio }
|
|
10
10
|
// no match → { image_size }
|
|
11
|
-
|
|
11
|
+
if (typeof size === 'string') {
|
|
12
|
+
const match = size.match(/^(\d+:\d+)(?:_(.+))?$/)
|
|
13
|
+
if (match) {
|
|
14
|
+
return {
|
|
15
|
+
aspect_ratio: match[1],
|
|
16
|
+
...(match[2] ? { resolution: match[2] } : {}),
|
|
17
|
+
} as FalModelImageSizeInput<TModel>
|
|
18
|
+
}
|
|
19
|
+
}
|
|
12
20
|
|
|
13
21
|
return {
|
|
14
22
|
image_size: size,
|
|
15
|
-
|
|
16
|
-
...(match?.[2] && {
|
|
17
|
-
resolution: match?.[2],
|
|
18
|
-
}),
|
|
19
|
-
}
|
|
23
|
+
} as FalModelImageSizeInput<TModel>
|
|
20
24
|
}
|
package/src/index.ts
CHANGED
|
@@ -10,6 +10,27 @@ export { FalImageAdapter, falImage } from './adapters/image'
|
|
|
10
10
|
|
|
11
11
|
export { FalVideoAdapter, falVideo } from './adapters/video'
|
|
12
12
|
|
|
13
|
+
// ============================================================================
|
|
14
|
+
// Speech Adapter (TTS)
|
|
15
|
+
// ============================================================================
|
|
16
|
+
|
|
17
|
+
export { FalSpeechAdapter, falSpeech } from './adapters/speech'
|
|
18
|
+
|
|
19
|
+
// ============================================================================
|
|
20
|
+
// Transcription Adapter (STT)
|
|
21
|
+
// ============================================================================
|
|
22
|
+
|
|
23
|
+
export {
|
|
24
|
+
FalTranscriptionAdapter,
|
|
25
|
+
falTranscription,
|
|
26
|
+
} from './adapters/transcription'
|
|
27
|
+
|
|
28
|
+
// ============================================================================
|
|
29
|
+
// Audio Adapter (Music/Sound Generation)
|
|
30
|
+
// ============================================================================
|
|
31
|
+
|
|
32
|
+
export { FalAudioAdapter, falAudio } from './adapters/audio'
|
|
33
|
+
|
|
13
34
|
// ============================================================================
|
|
14
35
|
// Model Types (from fal.ai's type system)
|
|
15
36
|
// ============================================================================
|
|
@@ -17,6 +38,9 @@ export { FalVideoAdapter, falVideo } from './adapters/video'
|
|
|
17
38
|
export {
|
|
18
39
|
type FalImageProviderOptions,
|
|
19
40
|
type FalVideoProviderOptions,
|
|
41
|
+
type FalSpeechProviderOptions,
|
|
42
|
+
type FalTranscriptionProviderOptions,
|
|
43
|
+
type FalAudioProviderOptions,
|
|
20
44
|
type FalModel,
|
|
21
45
|
type FalModelInput,
|
|
22
46
|
type FalModelOutput,
|
package/src/model-meta.ts
CHANGED
|
@@ -119,3 +119,30 @@ export type FalVideoProviderOptions<TModel extends string> =
|
|
|
119
119
|
TModel extends keyof EndpointTypeMap
|
|
120
120
|
? Omit<FalModelInput<TModel>, 'prompt'>
|
|
121
121
|
: Record<string, any>
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Provider options for TTS, excluding fields TanStack AI handles.
|
|
125
|
+
* Use this for the `modelOptions` parameter in speech generation.
|
|
126
|
+
*/
|
|
127
|
+
export type FalSpeechProviderOptions<TModel extends string> = Omit<
|
|
128
|
+
FalModelInput<TModel>,
|
|
129
|
+
'prompt' | 'text'
|
|
130
|
+
>
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Provider options for transcription, excluding fields TanStack AI handles.
|
|
134
|
+
* Use this for the `modelOptions` parameter in transcription.
|
|
135
|
+
*/
|
|
136
|
+
export type FalTranscriptionProviderOptions<TModel extends string> = Omit<
|
|
137
|
+
FalModelInput<TModel>,
|
|
138
|
+
'audio_url'
|
|
139
|
+
>
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Provider options for audio generation, excluding fields TanStack AI handles.
|
|
143
|
+
* Use this for the `modelOptions` parameter in audio generation.
|
|
144
|
+
*/
|
|
145
|
+
export type FalAudioProviderOptions<TModel extends string> = Omit<
|
|
146
|
+
FalModelInput<TModel>,
|
|
147
|
+
'prompt'
|
|
148
|
+
>
|
package/src/utils/client.ts
CHANGED
|
@@ -40,21 +40,129 @@ export function getFalApiKeyFromEnv(): string {
|
|
|
40
40
|
}
|
|
41
41
|
|
|
42
42
|
export function configureFalClient(config?: FalClientConfig): void {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
})
|
|
47
|
-
}
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
43
|
+
const apiKey = config?.apiKey ?? getFalApiKeyFromEnv()
|
|
44
|
+
fal.config({
|
|
45
|
+
credentials: apiKey,
|
|
46
|
+
...(config?.proxyUrl ? { proxyUrl: config.proxyUrl } : {}),
|
|
47
|
+
})
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function generateId(prefix: string): string {
|
|
51
|
+
return `${prefix}-${Date.now()}-${Math.random().toString(36).substring(2)}`
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Extract a safe file extension from a URL. Strips query strings, URL
|
|
56
|
+
* fragments, and any trailing slashes, and only returns the extension when
|
|
57
|
+
* it looks like a real one (2-5 alphanumeric chars). Returns undefined
|
|
58
|
+
* otherwise so callers can fall back to a default.
|
|
59
|
+
*/
|
|
60
|
+
export function extractUrlExtension(url: string): string | undefined {
|
|
61
|
+
// Parse via URL when possible so we only look at the pathname and never
|
|
62
|
+
// mistake a TLD (e.g. the `.com` in `https://x.com/`) for a file extension.
|
|
63
|
+
let pathname: string
|
|
64
|
+
try {
|
|
65
|
+
const parsed = new URL(url)
|
|
66
|
+
pathname = parsed.pathname
|
|
67
|
+
} catch {
|
|
68
|
+
// Fall back to treating the input as a raw path when URL parsing fails
|
|
69
|
+
// (e.g. the caller passed a bare path). Still strip ?query and #fragment.
|
|
70
|
+
pathname = url.split('?')[0]!.split('#')[0]!
|
|
71
|
+
}
|
|
72
|
+
// Drop trailing slashes so `/path/audio.mp3/` still yields `mp3`.
|
|
73
|
+
const normalized = pathname.replace(/\/+$/, '')
|
|
74
|
+
// Require at least one `/` — otherwise we're looking at an empty pathname
|
|
75
|
+
// (bare-host URLs like `https://x.com/` land here after stripping the
|
|
76
|
+
// trailing slash).
|
|
77
|
+
if (!normalized.includes('/')) return undefined
|
|
78
|
+
const lastSegment = normalized.split('/').pop()
|
|
79
|
+
if (!lastSegment) return undefined
|
|
80
|
+
const extension = lastSegment.split('.').pop()
|
|
81
|
+
if (!extension || extension === lastSegment) return undefined
|
|
82
|
+
return /^[a-z0-9]{2,5}$/i.test(extension) ? extension : undefined
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Derive a reasonable audio content-type. Prefers the explicit MIME (stripped
|
|
87
|
+
* of parameters), then an extension-based lookup for common audio formats,
|
|
88
|
+
* otherwise falls back to audio/mpeg — fal URLs virtually always serve mp3.
|
|
89
|
+
*/
|
|
90
|
+
export function deriveAudioContentType(
|
|
91
|
+
explicitContentType: string | undefined,
|
|
92
|
+
url: string,
|
|
93
|
+
): string {
|
|
94
|
+
const stripped = explicitContentType?.split(';')[0]?.trim()
|
|
95
|
+
if (stripped) return stripped
|
|
96
|
+
|
|
97
|
+
const ext = extractUrlExtension(url)?.toLowerCase()
|
|
98
|
+
switch (ext) {
|
|
99
|
+
case 'mp3':
|
|
100
|
+
return 'audio/mpeg'
|
|
101
|
+
case 'wav':
|
|
102
|
+
return 'audio/wav'
|
|
103
|
+
case 'ogg':
|
|
104
|
+
case 'oga':
|
|
105
|
+
return 'audio/ogg'
|
|
106
|
+
case 'flac':
|
|
107
|
+
return 'audio/flac'
|
|
108
|
+
case 'aac':
|
|
109
|
+
return 'audio/aac'
|
|
110
|
+
case 'm4a':
|
|
111
|
+
case 'mp4':
|
|
112
|
+
return 'audio/mp4'
|
|
113
|
+
case 'webm':
|
|
114
|
+
return 'audio/webm'
|
|
115
|
+
default:
|
|
116
|
+
return 'audio/mpeg'
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Decode a `data:` URL into a Blob so fal-client can auto-upload it via
|
|
122
|
+
* `fal.storage.upload`. fal's inference API rejects data URLs with a 422
|
|
123
|
+
* "Unsupported data URL", so we convert them before handing them off.
|
|
124
|
+
*
|
|
125
|
+
* Supports both base64 and URL-encoded data URLs. Returns `undefined` for
|
|
126
|
+
* anything that isn't a data URL, so callers can fall through to other
|
|
127
|
+
* handling (http URLs are passed to fal as-is).
|
|
128
|
+
*/
|
|
129
|
+
export function dataUrlToBlob(value: string): Blob | undefined {
|
|
130
|
+
if (!value.startsWith('data:')) return undefined
|
|
131
|
+
const commaIndex = value.indexOf(',')
|
|
132
|
+
if (commaIndex === -1) return undefined
|
|
133
|
+
|
|
134
|
+
const header = value.slice(5, commaIndex)
|
|
135
|
+
const payload = value.slice(commaIndex + 1)
|
|
136
|
+
const isBase64 = /;base64$/i.test(header)
|
|
137
|
+
const mimeType = header.split(';')[0] || 'application/octet-stream'
|
|
138
|
+
|
|
139
|
+
if (isBase64) {
|
|
140
|
+
const binary = atob(payload)
|
|
141
|
+
const bytes = new Uint8Array(binary.length)
|
|
142
|
+
for (let i = 0; i < binary.length; i += 1) {
|
|
143
|
+
bytes[i] = binary.charCodeAt(i)
|
|
51
144
|
}
|
|
52
|
-
|
|
53
|
-
credentials: apiKey,
|
|
54
|
-
})
|
|
145
|
+
return new Blob([bytes], { type: mimeType })
|
|
55
146
|
}
|
|
147
|
+
|
|
148
|
+
return new Blob([decodeURIComponent(payload)], { type: mimeType })
|
|
56
149
|
}
|
|
57
150
|
|
|
58
|
-
|
|
59
|
-
|
|
151
|
+
/**
|
|
152
|
+
* Convert an ArrayBuffer to base64 in a cross-runtime way.
|
|
153
|
+
*
|
|
154
|
+
* The naive `btoa(String.fromCharCode(...bytes))` form blows up V8's argument
|
|
155
|
+
* limit (~65k) on realistic audio payloads, so we either use `Buffer` (Node /
|
|
156
|
+
* Bun) or walk the byte array in a single loop (browser).
|
|
157
|
+
*/
|
|
158
|
+
export function arrayBufferToBase64(bytes: ArrayBuffer): string {
|
|
159
|
+
if (typeof Buffer !== 'undefined' && typeof Buffer.from === 'function') {
|
|
160
|
+
return Buffer.from(bytes).toString('base64')
|
|
161
|
+
}
|
|
162
|
+
const view = new Uint8Array(bytes)
|
|
163
|
+
let binary = ''
|
|
164
|
+
for (let i = 0; i < view.byteLength; i += 1) {
|
|
165
|
+
binary += String.fromCharCode(view[i]!)
|
|
166
|
+
}
|
|
167
|
+
return btoa(binary)
|
|
60
168
|
}
|