runonweb 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -0
- package/package.json +126 -0
- package/src/caption/index.ts +310 -0
- package/src/clean/index.ts +263 -0
- package/src/core/cache.ts +248 -0
- package/src/core/device.ts +34 -0
- package/src/core/index.ts +18 -0
- package/src/core/pipeline.ts +108 -0
- package/src/core/progress.ts +20 -0
- package/src/depth/index.ts +127 -0
- package/src/detect/index.ts +274 -0
- package/src/embed/index.ts +172 -0
- package/src/emoji/index.ts +132 -0
- package/src/image/engine.d.ts +57 -0
- package/src/image/engine.js +14090 -0
- package/src/image/index.ts +230 -0
- package/src/image/sizes.ts +28 -0
- package/src/ocr/index.ts +416 -0
- package/src/ocr/sizes.ts +12 -0
- package/src/remove-bg/index.ts +170 -0
- package/src/stt/index.ts +320 -0
- package/src/translate/bergamot.ts +280 -0
- package/src/translate/index.ts +240 -0
- package/src/translate/registry.ts +133 -0
- package/src/translate/worker.ts +181 -0
- package/src/tts/index.ts +220 -0
- package/src/tts/kitten.ts +280 -0
- package/src/tts/kokoro.ts +140 -0
- package/src/tts/phonemes.ts +150 -0
- package/src/tts/sizes.ts +30 -0
- package/src/tts/split.ts +31 -0
- package/src/tts/supertonic.ts +126 -0
- package/src/tts/types.ts +11 -0
- package/src/tts/voices.ts +131 -0
- package/src/tts/wav.ts +52 -0
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
import type { Device, ProgressCallback, ProgressInfo } from '../core/index.ts'
|
|
2
|
+
import { imageToPipelineInput, loadPipeline } from '../core/pipeline.ts'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Default: BEN2 (MIT, 2025, ~219 MB fp16). General background eraser: hair, objects,
|
|
6
|
+
* hard edges. The only ONNX weight is fp16.
|
|
7
|
+
*
|
|
8
|
+
* BEN2 fuses LayerNormalization with an fp16 activation and fp32 scale, bias, and
|
|
9
|
+
* output. onnxruntime-web 1.30.0 through 1.31.0-dev.20260918 assigns `vec4<f16>` to
|
|
10
|
+
* `vec4<f32>` storage, so the first WebGPU `OrtRun` fails with
|
|
11
|
+
* `Invalid ShaderModule "LayerNorm"`. The workspace pins Transformers.js to
|
|
12
|
+
* onnxruntime-web 1.29.0, the last release without the bug, until one ships
|
|
13
|
+
* onnxruntime#32629. https://github.com/microsoft/onnxruntime/issues/32627
|
|
14
|
+
*/
|
|
15
|
+
const DEFAULT_MODEL = 'onnx-community/BEN2-ONNX'
|
|
16
|
+
|
|
17
|
+
/** Shader compile failures surface at run time, after the WebGPU session has loaded. */
|
|
18
|
+
function isWebGpuShaderError(err: unknown): boolean {
|
|
19
|
+
const message = err instanceof Error ? err.message : String(err)
|
|
20
|
+
return /Invalid ShaderModule|LayerNorm|failed to call OrtRun/i.test(message)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export type RemoveBgOptions = {
|
|
24
|
+
/** Hugging Face model id. Defaults to BEN2. */
|
|
25
|
+
model?: string
|
|
26
|
+
/** Inference device. Defaults to `auto` (WebGPU when available, else WASM). */
|
|
27
|
+
device?: Device
|
|
28
|
+
/** Called while model files download / load. */
|
|
29
|
+
onProgress?: ProgressCallback
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export type RemoveBgImageInput =
|
|
33
|
+
| Blob
|
|
34
|
+
| File
|
|
35
|
+
| string
|
|
36
|
+
| HTMLImageElement
|
|
37
|
+
| ImageData
|
|
38
|
+
| HTMLCanvasElement
|
|
39
|
+
|
|
40
|
+
type BackgroundRemovalPipeline = {
|
|
41
|
+
(
|
|
42
|
+
images: Array<unknown>,
|
|
43
|
+
options?: Record<string, unknown>
|
|
44
|
+
): Promise<Array<{ toBlob: () => Promise<Blob>; toCanvas: () => HTMLCanvasElement }>>
|
|
45
|
+
dispose?: () => Promise<void>
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Background removal that runs entirely in the browser.
|
|
50
|
+
*
|
|
51
|
+
* @example
|
|
52
|
+
* ```ts
|
|
53
|
+
* import { RemoveBackground } from 'runonweb/remove-bg'
|
|
54
|
+
*
|
|
55
|
+
* const remover = new RemoveBackground()
|
|
56
|
+
* await remover.load()
|
|
57
|
+
* const png = await remover.remove(imageFile) // PNG Blob with alpha
|
|
58
|
+
* ```
|
|
59
|
+
*/
|
|
60
|
+
export class RemoveBackground {
|
|
61
|
+
#model: string
|
|
62
|
+
#device: Device
|
|
63
|
+
#onProgress?: ProgressCallback
|
|
64
|
+
#pipe: BackgroundRemovalPipeline | null = null
|
|
65
|
+
#loading: Promise<void> | null = null
|
|
66
|
+
#resolvedDevice: 'webgpu' | 'wasm' | null = null
|
|
67
|
+
|
|
68
|
+
constructor(options: RemoveBgOptions = {}) {
|
|
69
|
+
this.#model = options.model ?? DEFAULT_MODEL
|
|
70
|
+
this.#device = options.device ?? 'auto'
|
|
71
|
+
this.#onProgress = options.onProgress
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Device actually used after `load()`. */
|
|
75
|
+
get device(): 'webgpu' | 'wasm' | null {
|
|
76
|
+
return this.#resolvedDevice
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Download and initialize the model. Safe to call multiple times. */
|
|
80
|
+
async load(): Promise<void> {
|
|
81
|
+
if (this.#pipe) return
|
|
82
|
+
if (this.#loading) return this.#loading
|
|
83
|
+
|
|
84
|
+
this.#loading = (async () => {
|
|
85
|
+
const load = (device: Device) =>
|
|
86
|
+
loadPipeline({
|
|
87
|
+
task: 'background-removal',
|
|
88
|
+
model: this.#model,
|
|
89
|
+
device,
|
|
90
|
+
// BEN2 ships fp16 only (~219 MB).
|
|
91
|
+
dtype: 'fp16',
|
|
92
|
+
onProgress: this.#onProgress,
|
|
93
|
+
})
|
|
94
|
+
|
|
95
|
+
try {
|
|
96
|
+
const { pipe, device } = await load(this.#device)
|
|
97
|
+
this.#pipe = pipe as unknown as BackgroundRemovalPipeline
|
|
98
|
+
this.#resolvedDevice = device
|
|
99
|
+
} catch (err) {
|
|
100
|
+
if (this.#device === 'wasm') throw err
|
|
101
|
+
const { pipe, device } = await load('wasm')
|
|
102
|
+
this.#pipe = pipe as unknown as BackgroundRemovalPipeline
|
|
103
|
+
this.#resolvedDevice = device
|
|
104
|
+
}
|
|
105
|
+
})()
|
|
106
|
+
|
|
107
|
+
try {
|
|
108
|
+
await this.#loading
|
|
109
|
+
} finally {
|
|
110
|
+
this.#loading = null
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Remove the background from an image.
|
|
116
|
+
* Returns a PNG `Blob` with an alpha channel.
|
|
117
|
+
*/
|
|
118
|
+
async remove(image: RemoveBgImageInput): Promise<Blob> {
|
|
119
|
+
try {
|
|
120
|
+
return await this.#remove(image)
|
|
121
|
+
} catch (err) {
|
|
122
|
+
// A WebGPU session can load and still die in a shader at run time. Retry once on WASM.
|
|
123
|
+
if (this.#resolvedDevice !== 'webgpu' || !isWebGpuShaderError(err)) throw err
|
|
124
|
+
this.dispose()
|
|
125
|
+
this.#device = 'wasm'
|
|
126
|
+
return await this.#remove(image)
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
async #remove(image: RemoveBgImageInput): Promise<Blob> {
|
|
131
|
+
await this.load()
|
|
132
|
+
if (!this.#pipe) throw new Error('RemoveBackground model failed to load')
|
|
133
|
+
|
|
134
|
+
this.#onProgress?.({ status: 'processing' })
|
|
135
|
+
|
|
136
|
+
const input = await imageToPipelineInput(image)
|
|
137
|
+
const output = await this.#pipe([input])
|
|
138
|
+
const result = output[0]
|
|
139
|
+
if (!result) throw new Error('Background removal produced no output')
|
|
140
|
+
|
|
141
|
+
const blob = await result.toBlob()
|
|
142
|
+
this.#onProgress?.({ status: 'done' })
|
|
143
|
+
return blob
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** Release model resources. */
|
|
147
|
+
dispose(): void {
|
|
148
|
+
const pipe = this.#pipe
|
|
149
|
+
this.#pipe = null
|
|
150
|
+
this.#resolvedDevice = null
|
|
151
|
+
void pipe?.dispose?.()
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* One-shot helper: load model, remove background, dispose.
|
|
157
|
+
*/
|
|
158
|
+
export async function removeBackground(
|
|
159
|
+
image: RemoveBgImageInput,
|
|
160
|
+
options?: RemoveBgOptions
|
|
161
|
+
): Promise<Blob> {
|
|
162
|
+
const remover = new RemoveBackground(options)
|
|
163
|
+
try {
|
|
164
|
+
return await remover.remove(image)
|
|
165
|
+
} finally {
|
|
166
|
+
remover.dispose()
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
export type { ProgressInfo, Device }
|
package/src/stt/index.ts
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
import type { Device, ProgressCallback, ProgressInfo } from '../core/index.ts'
|
|
2
|
+
import { loadPipeline } from '../core/pipeline.ts'
|
|
3
|
+
|
|
4
|
+
const DEFAULT_MODEL = 'onnx-community/whisper-tiny.en'
|
|
5
|
+
const SAMPLE_RATE = 16_000
|
|
6
|
+
|
|
7
|
+
export type STTOptions = {
|
|
8
|
+
/** Hugging Face model id. Defaults to a small English Whisper. */
|
|
9
|
+
model?: string
|
|
10
|
+
/** Inference device. Defaults to `auto` (WebGPU when available, else WASM). */
|
|
11
|
+
device?: Device
|
|
12
|
+
/** Language code for multilingual models (e.g. `"es"`, `"en"`). */
|
|
13
|
+
language?: string
|
|
14
|
+
/** Called while model files download / load. */
|
|
15
|
+
onProgress?: ProgressCallback
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export type STTChunk = {
|
|
19
|
+
text: string
|
|
20
|
+
start: number
|
|
21
|
+
end: number
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export type STTResult = {
|
|
25
|
+
text: string
|
|
26
|
+
chunks: STTChunk[]
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export type TranscribeOptions = {
|
|
30
|
+
/** Called with the transcript so far as Whisper emits words. */
|
|
31
|
+
onPartial?: (text: string) => void
|
|
32
|
+
/** Per-call language for multilingual models (e.g. `"es"`, `"ja"`). Falls back to the constructor value. */
|
|
33
|
+
language?: string
|
|
34
|
+
/**
|
|
35
|
+
* Whisper task. Default `transcribe` keeps the source language.
|
|
36
|
+
* `translate` turns speech into English. Never the default here.
|
|
37
|
+
*/
|
|
38
|
+
task?: 'transcribe' | 'translate'
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export type STTAudioInput = Blob | File | string | Float32Array
|
|
42
|
+
|
|
43
|
+
type RawChunk = { text: string; timestamp: [number, number | null] }
|
|
44
|
+
type RawResult = { text: string; chunks?: RawChunk[] }
|
|
45
|
+
|
|
46
|
+
type ASRPipeline = {
|
|
47
|
+
(audio: Float32Array | string, options?: Record<string, unknown>): Promise<RawResult | RawResult[]>
|
|
48
|
+
tokenizer?: unknown
|
|
49
|
+
dispose?: () => Promise<void>
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Speech-to-text that runs entirely in the browser (Whisper via Transformers.js).
|
|
54
|
+
*
|
|
55
|
+
* @example
|
|
56
|
+
* ```ts
|
|
57
|
+
* import { SpeechToText } from 'runonweb/stt'
|
|
58
|
+
*
|
|
59
|
+
* const stt = new SpeechToText()
|
|
60
|
+
* await stt.load()
|
|
61
|
+
* const { text } = await stt.transcribe(audioBlob)
|
|
62
|
+
* console.log(text)
|
|
63
|
+
* ```
|
|
64
|
+
*/
|
|
65
|
+
export class SpeechToText {
|
|
66
|
+
#model: string
|
|
67
|
+
#device: Device
|
|
68
|
+
#language?: string
|
|
69
|
+
#onProgress?: ProgressCallback
|
|
70
|
+
#pipe: ASRPipeline | null = null
|
|
71
|
+
#loading: Promise<void> | null = null
|
|
72
|
+
#resolvedDevice: 'webgpu' | 'wasm' | null = null
|
|
73
|
+
|
|
74
|
+
constructor(options: STTOptions = {}) {
|
|
75
|
+
this.#model = options.model ?? DEFAULT_MODEL
|
|
76
|
+
this.#device = options.device ?? 'auto'
|
|
77
|
+
this.#language = options.language
|
|
78
|
+
this.#onProgress = options.onProgress
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Device actually used after `load()`. */
|
|
82
|
+
get device(): 'webgpu' | 'wasm' | null {
|
|
83
|
+
return this.#resolvedDevice
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Download and initialize the model. Safe to call multiple times. */
|
|
87
|
+
async load(): Promise<void> {
|
|
88
|
+
if (this.#pipe) return
|
|
89
|
+
if (this.#loading) return this.#loading
|
|
90
|
+
|
|
91
|
+
this.#loading = (async () => {
|
|
92
|
+
const { pipe, device } = await loadPipeline({
|
|
93
|
+
task: 'automatic-speech-recognition',
|
|
94
|
+
model: this.#model,
|
|
95
|
+
device: this.#device,
|
|
96
|
+
onProgress: this.#onProgress,
|
|
97
|
+
})
|
|
98
|
+
this.#pipe = pipe as unknown as ASRPipeline
|
|
99
|
+
this.#resolvedDevice = device
|
|
100
|
+
})()
|
|
101
|
+
|
|
102
|
+
try {
|
|
103
|
+
await this.#loading
|
|
104
|
+
} finally {
|
|
105
|
+
this.#loading = null
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Transcribe audio. Accepts a Blob/File, a URL string, or raw Float32Array samples (16 kHz mono).
|
|
111
|
+
* Pass `onPartial` to receive words as they are generated.
|
|
112
|
+
*/
|
|
113
|
+
async transcribe(audio: STTAudioInput, options?: TranscribeOptions): Promise<STTResult> {
|
|
114
|
+
await this.load()
|
|
115
|
+
if (!this.#pipe) throw new Error('SpeechToText model failed to load')
|
|
116
|
+
|
|
117
|
+
this.#onProgress?.({ status: 'transcribing' })
|
|
118
|
+
|
|
119
|
+
const input = await prepareAudio(audio)
|
|
120
|
+
const duration = input instanceof Float32Array ? input.length / SAMPLE_RATE : 0
|
|
121
|
+
const onPartial = options?.onPartial
|
|
122
|
+
let streamed = ''
|
|
123
|
+
|
|
124
|
+
const base: Record<string, unknown> = {
|
|
125
|
+
chunk_length_s: 30,
|
|
126
|
+
stride_length_s: 5,
|
|
127
|
+
// Always transcribe unless asked otherwise. Multilingual Whisper will
|
|
128
|
+
// otherwise slip into "translate to English" when language is unknown.
|
|
129
|
+
task: options?.task ?? 'transcribe',
|
|
130
|
+
// Word-timestamp mode disables timestamp tokens. Without this, decode
|
|
131
|
+
// throws "Whisper did not predict an ending timestamp" and the retry
|
|
132
|
+
// can come back empty after the streamer already showed text.
|
|
133
|
+
force_full_sequences: false,
|
|
134
|
+
}
|
|
135
|
+
const language = options?.language ?? this.#language
|
|
136
|
+
if (language) base.language = language
|
|
137
|
+
|
|
138
|
+
const run = async (returnTimestamps: true | 'word', stream: boolean) => {
|
|
139
|
+
const streamer =
|
|
140
|
+
stream && onPartial ? await createStreamer(this.#pipe!, (text) => {
|
|
141
|
+
streamed = text
|
|
142
|
+
onPartial(text)
|
|
143
|
+
}) : undefined
|
|
144
|
+
return this.#pipe!(input, {
|
|
145
|
+
...base,
|
|
146
|
+
return_timestamps: returnTimestamps,
|
|
147
|
+
...(streamer ? { streamer } : {}),
|
|
148
|
+
})
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
let raw: RawResult | RawResult[]
|
|
152
|
+
try {
|
|
153
|
+
// Segment timestamps + streamer is the reliable path. Word timestamps
|
|
154
|
+
// need a different generate() return shape and often fail once a streamer
|
|
155
|
+
// is attached, wiping the final text after a good partial stream.
|
|
156
|
+
raw = await run(true, true)
|
|
157
|
+
} catch {
|
|
158
|
+
raw = await run(true, false)
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
const parsed = normalizeRaw(raw)
|
|
162
|
+
const text = parsed.text || streamed
|
|
163
|
+
const chunks = explodeWords(parsed.chunks.length ? parsed.chunks : wordsFromText(text, duration), duration)
|
|
164
|
+
|
|
165
|
+
this.#onProgress?.({ status: 'done' })
|
|
166
|
+
return { text, chunks }
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/** Release model resources. */
|
|
170
|
+
dispose(): void {
|
|
171
|
+
const pipe = this.#pipe
|
|
172
|
+
this.#pipe = null
|
|
173
|
+
this.#resolvedDevice = null
|
|
174
|
+
void pipe?.dispose?.()
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* One-shot helper: load model, transcribe, dispose.
|
|
180
|
+
*/
|
|
181
|
+
export async function transcribe(
|
|
182
|
+
audio: STTAudioInput,
|
|
183
|
+
options?: STTOptions & TranscribeOptions
|
|
184
|
+
): Promise<STTResult> {
|
|
185
|
+
const stt = new SpeechToText(options)
|
|
186
|
+
try {
|
|
187
|
+
return await stt.transcribe(audio, options)
|
|
188
|
+
} finally {
|
|
189
|
+
stt.dispose()
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
async function createStreamer(pipe: ASRPipeline, onPartial: (text: string) => void) {
|
|
194
|
+
const { WhisperTextStreamer } = await import('@huggingface/transformers')
|
|
195
|
+
let acc = ''
|
|
196
|
+
return new WhisperTextStreamer(pipe.tokenizer as never, {
|
|
197
|
+
skip_prompt: true,
|
|
198
|
+
skip_special_tokens: true,
|
|
199
|
+
callback_function: (piece: string) => {
|
|
200
|
+
acc += piece
|
|
201
|
+
const next = acc.replace(/\s+/g, ' ').trim()
|
|
202
|
+
if (next) onPartial(next)
|
|
203
|
+
},
|
|
204
|
+
})
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function normalizeRaw(raw: RawResult | RawResult[]): { text: string; chunks: STTChunk[] } {
|
|
208
|
+
const items = Array.isArray(raw) ? raw : [raw]
|
|
209
|
+
const text = items
|
|
210
|
+
.map((item) => item.text ?? '')
|
|
211
|
+
.join(' ')
|
|
212
|
+
.replace(/\s+/g, ' ')
|
|
213
|
+
.trim()
|
|
214
|
+
return { text, chunks: items.flatMap((item) => parseChunks(item.chunks)) }
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function parseChunks(raw?: RawChunk[]): STTChunk[] {
|
|
218
|
+
if (!raw?.length) return []
|
|
219
|
+
const chunks: STTChunk[] = []
|
|
220
|
+
for (const item of raw) {
|
|
221
|
+
const text = (item.text ?? '').replace(/\s+/g, ' ')
|
|
222
|
+
if (!text.trim()) continue
|
|
223
|
+
const start = item.timestamp?.[0] ?? chunks.at(-1)?.end ?? 0
|
|
224
|
+
const end = item.timestamp?.[1] ?? start
|
|
225
|
+
chunks.push({ text, start, end: end < start ? start : end })
|
|
226
|
+
}
|
|
227
|
+
return chunks
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
function wordsFromText(text: string, duration: number): STTChunk[] {
|
|
231
|
+
const parts = text.trim().split(/\s+/).filter(Boolean)
|
|
232
|
+
if (!parts.length) return []
|
|
233
|
+
const span = Math.max(0.04, (duration || parts.length * 0.35) / parts.length)
|
|
234
|
+
return parts.map((word, i) => ({
|
|
235
|
+
text: word,
|
|
236
|
+
start: i * span,
|
|
237
|
+
end: (i + 1) * span,
|
|
238
|
+
}))
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/** Split segment chunks into words and align timestamps to the audio duration. */
|
|
242
|
+
function explodeWords(raw: STTChunk[], duration: number): STTChunk[] {
|
|
243
|
+
const words: STTChunk[] = []
|
|
244
|
+
for (const chunk of raw) {
|
|
245
|
+
const parts = chunk.text.trim().split(/\s+/).filter(Boolean)
|
|
246
|
+
if (!parts.length) continue
|
|
247
|
+
if (parts.length === 1) {
|
|
248
|
+
words.push({ text: parts[0]!, start: chunk.start, end: Math.max(chunk.end, chunk.start) })
|
|
249
|
+
continue
|
|
250
|
+
}
|
|
251
|
+
const start = chunk.start
|
|
252
|
+
const end = Math.max(chunk.end, chunk.start)
|
|
253
|
+
const span = Math.max(0.04, (end - start || parts.length * 0.35) / parts.length)
|
|
254
|
+
parts.forEach((word, i) => {
|
|
255
|
+
words.push({
|
|
256
|
+
text: word,
|
|
257
|
+
start: start + i * span,
|
|
258
|
+
end: start + (i + 1) * span,
|
|
259
|
+
})
|
|
260
|
+
})
|
|
261
|
+
}
|
|
262
|
+
return alignToDuration(words, duration)
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
function alignToDuration(chunks: STTChunk[], duration: number): STTChunk[] {
|
|
266
|
+
if (!chunks.length) return chunks
|
|
267
|
+
const maxEnd = Math.max(...chunks.map((c) => Math.max(c.end, c.start)))
|
|
268
|
+
if (maxEnd <= 0.05 && duration > 0) return wordsFromText(chunks.map((c) => c.text).join(' '), duration)
|
|
269
|
+
if (duration > 0 && (maxEnd > duration * 1.2 || maxEnd < duration * 0.55)) {
|
|
270
|
+
const scale = duration / maxEnd
|
|
271
|
+
return chunks.map((c) => ({ ...c, start: c.start * scale, end: Math.max(c.end, c.start) * scale }))
|
|
272
|
+
}
|
|
273
|
+
return chunks
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
export type { ProgressInfo, Device }
|
|
277
|
+
|
|
278
|
+
async function prepareAudio(audio: STTAudioInput): Promise<Float32Array | string> {
|
|
279
|
+
if (typeof audio === 'string') return audio
|
|
280
|
+
if (audio instanceof Float32Array) return audio
|
|
281
|
+
|
|
282
|
+
const arrayBuffer = await audio.arrayBuffer()
|
|
283
|
+
const AudioCtx =
|
|
284
|
+
globalThis.AudioContext ??
|
|
285
|
+
(globalThis as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext
|
|
286
|
+
const ctx = new AudioCtx()
|
|
287
|
+
try {
|
|
288
|
+
const decoded = await ctx.decodeAudioData(arrayBuffer.slice(0))
|
|
289
|
+
const mono = mixMono(decoded)
|
|
290
|
+
return resample(mono, decoded.sampleRate, SAMPLE_RATE)
|
|
291
|
+
} finally {
|
|
292
|
+
await ctx.close()
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function mixMono(buffer: AudioBuffer): Float32Array {
|
|
297
|
+
if (buffer.numberOfChannels === 1) return new Float32Array(buffer.getChannelData(0))
|
|
298
|
+
const len = buffer.length
|
|
299
|
+
const out = new Float32Array(len)
|
|
300
|
+
const scale = buffer.numberOfChannels === 2 ? Math.SQRT1_2 : 1 / buffer.numberOfChannels
|
|
301
|
+
for (let c = 0; c < buffer.numberOfChannels; c++) {
|
|
302
|
+
const channel = buffer.getChannelData(c)
|
|
303
|
+
for (let i = 0; i < len; i++) out[i] += channel[i]! * scale
|
|
304
|
+
}
|
|
305
|
+
return out
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
function resample(input: Float32Array, from: number, to: number): Float32Array {
|
|
309
|
+
if (from === to) return input
|
|
310
|
+
const ratio = from / to
|
|
311
|
+
const out = new Float32Array(Math.round(input.length / ratio))
|
|
312
|
+
for (let i = 0; i < out.length; i++) {
|
|
313
|
+
const x = i * ratio
|
|
314
|
+
const i0 = Math.min(Math.floor(x), input.length - 1)
|
|
315
|
+
const i1 = Math.min(i0 + 1, input.length - 1)
|
|
316
|
+
const t = x - i0
|
|
317
|
+
out[i] = input[i0]! * (1 - t) + input[i1]! * t
|
|
318
|
+
}
|
|
319
|
+
return out
|
|
320
|
+
}
|