runonweb 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,170 @@
1
+ import type { Device, ProgressCallback, ProgressInfo } from '../core/index.ts'
2
+ import { imageToPipelineInput, loadPipeline } from '../core/pipeline.ts'
3
+
4
+ /**
5
+ * Default: BEN2 (MIT, 2025, ~219 MB fp16). General background eraser: hair, objects,
6
+ * hard edges. The only ONNX weight is fp16.
7
+ *
8
+ * BEN2 fuses LayerNormalization with an fp16 activation and fp32 scale, bias, and
9
+ * output. onnxruntime-web 1.30.0 through 1.31.0-dev.20260918 assigns `vec4<f16>` to
10
+ * `vec4<f32>` storage, so the first WebGPU `OrtRun` fails with
11
+ * `Invalid ShaderModule "LayerNorm"`. The workspace pins Transformers.js to
12
+ * onnxruntime-web 1.29.0, the last release without the bug, until one ships
13
+ * onnxruntime#32629. https://github.com/microsoft/onnxruntime/issues/32627
14
+ */
15
+ const DEFAULT_MODEL = 'onnx-community/BEN2-ONNX'
16
+
17
+ /** Shader compile failures surface at run time, after the WebGPU session has loaded. */
18
+ function isWebGpuShaderError(err: unknown): boolean {
19
+ const message = err instanceof Error ? err.message : String(err)
20
+ return /Invalid ShaderModule|LayerNorm|failed to call OrtRun/i.test(message)
21
+ }
22
+
23
+ export type RemoveBgOptions = {
24
+ /** Hugging Face model id. Defaults to BEN2. */
25
+ model?: string
26
+ /** Inference device. Defaults to `auto` (WebGPU when available, else WASM). */
27
+ device?: Device
28
+ /** Called while model files download / load. */
29
+ onProgress?: ProgressCallback
30
+ }
31
+
32
+ export type RemoveBgImageInput =
33
+ | Blob
34
+ | File
35
+ | string
36
+ | HTMLImageElement
37
+ | ImageData
38
+ | HTMLCanvasElement
39
+
40
+ type BackgroundRemovalPipeline = {
41
+ (
42
+ images: Array<unknown>,
43
+ options?: Record<string, unknown>
44
+ ): Promise<Array<{ toBlob: () => Promise<Blob>; toCanvas: () => HTMLCanvasElement }>>
45
+ dispose?: () => Promise<void>
46
+ }
47
+
48
+ /**
49
+ * Background removal that runs entirely in the browser.
50
+ *
51
+ * @example
52
+ * ```ts
53
+ * import { RemoveBackground } from 'runonweb/remove-bg'
54
+ *
55
+ * const remover = new RemoveBackground()
56
+ * await remover.load()
57
+ * const png = await remover.remove(imageFile) // PNG Blob with alpha
58
+ * ```
59
+ */
60
+ export class RemoveBackground {
61
+ #model: string
62
+ #device: Device
63
+ #onProgress?: ProgressCallback
64
+ #pipe: BackgroundRemovalPipeline | null = null
65
+ #loading: Promise<void> | null = null
66
+ #resolvedDevice: 'webgpu' | 'wasm' | null = null
67
+
68
+ constructor(options: RemoveBgOptions = {}) {
69
+ this.#model = options.model ?? DEFAULT_MODEL
70
+ this.#device = options.device ?? 'auto'
71
+ this.#onProgress = options.onProgress
72
+ }
73
+
74
+ /** Device actually used after `load()`. */
75
+ get device(): 'webgpu' | 'wasm' | null {
76
+ return this.#resolvedDevice
77
+ }
78
+
79
+ /** Download and initialize the model. Safe to call multiple times. */
80
+ async load(): Promise<void> {
81
+ if (this.#pipe) return
82
+ if (this.#loading) return this.#loading
83
+
84
+ this.#loading = (async () => {
85
+ const load = (device: Device) =>
86
+ loadPipeline({
87
+ task: 'background-removal',
88
+ model: this.#model,
89
+ device,
90
+ // BEN2 ships fp16 only (~219 MB).
91
+ dtype: 'fp16',
92
+ onProgress: this.#onProgress,
93
+ })
94
+
95
+ try {
96
+ const { pipe, device } = await load(this.#device)
97
+ this.#pipe = pipe as unknown as BackgroundRemovalPipeline
98
+ this.#resolvedDevice = device
99
+ } catch (err) {
100
+ if (this.#device === 'wasm') throw err
101
+ const { pipe, device } = await load('wasm')
102
+ this.#pipe = pipe as unknown as BackgroundRemovalPipeline
103
+ this.#resolvedDevice = device
104
+ }
105
+ })()
106
+
107
+ try {
108
+ await this.#loading
109
+ } finally {
110
+ this.#loading = null
111
+ }
112
+ }
113
+
114
+ /**
115
+ * Remove the background from an image.
116
+ * Returns a PNG `Blob` with an alpha channel.
117
+ */
118
+ async remove(image: RemoveBgImageInput): Promise<Blob> {
119
+ try {
120
+ return await this.#remove(image)
121
+ } catch (err) {
122
+ // A WebGPU session can load and still die in a shader at run time. Retry once on WASM.
123
+ if (this.#resolvedDevice !== 'webgpu' || !isWebGpuShaderError(err)) throw err
124
+ this.dispose()
125
+ this.#device = 'wasm'
126
+ return await this.#remove(image)
127
+ }
128
+ }
129
+
130
+ async #remove(image: RemoveBgImageInput): Promise<Blob> {
131
+ await this.load()
132
+ if (!this.#pipe) throw new Error('RemoveBackground model failed to load')
133
+
134
+ this.#onProgress?.({ status: 'processing' })
135
+
136
+ const input = await imageToPipelineInput(image)
137
+ const output = await this.#pipe([input])
138
+ const result = output[0]
139
+ if (!result) throw new Error('Background removal produced no output')
140
+
141
+ const blob = await result.toBlob()
142
+ this.#onProgress?.({ status: 'done' })
143
+ return blob
144
+ }
145
+
146
+ /** Release model resources. */
147
+ dispose(): void {
148
+ const pipe = this.#pipe
149
+ this.#pipe = null
150
+ this.#resolvedDevice = null
151
+ void pipe?.dispose?.()
152
+ }
153
+ }
154
+
155
+ /**
156
+ * One-shot helper: load model, remove background, dispose.
157
+ */
158
+ export async function removeBackground(
159
+ image: RemoveBgImageInput,
160
+ options?: RemoveBgOptions
161
+ ): Promise<Blob> {
162
+ const remover = new RemoveBackground(options)
163
+ try {
164
+ return await remover.remove(image)
165
+ } finally {
166
+ remover.dispose()
167
+ }
168
+ }
169
+
170
+ export type { ProgressInfo, Device }
@@ -0,0 +1,320 @@
1
+ import type { Device, ProgressCallback, ProgressInfo } from '../core/index.ts'
2
+ import { loadPipeline } from '../core/pipeline.ts'
3
+
4
+ const DEFAULT_MODEL = 'onnx-community/whisper-tiny.en'
5
+ const SAMPLE_RATE = 16_000
6
+
7
+ export type STTOptions = {
8
+ /** Hugging Face model id. Defaults to a small English Whisper. */
9
+ model?: string
10
+ /** Inference device. Defaults to `auto` (WebGPU when available, else WASM). */
11
+ device?: Device
12
+ /** Language code for multilingual models (e.g. `"es"`, `"en"`). */
13
+ language?: string
14
+ /** Called while model files download / load. */
15
+ onProgress?: ProgressCallback
16
+ }
17
+
18
+ export type STTChunk = {
19
+ text: string
20
+ start: number
21
+ end: number
22
+ }
23
+
24
+ export type STTResult = {
25
+ text: string
26
+ chunks: STTChunk[]
27
+ }
28
+
29
+ export type TranscribeOptions = {
30
+ /** Called with the transcript so far as Whisper emits words. */
31
+ onPartial?: (text: string) => void
32
+ /** Per-call language for multilingual models (e.g. `"es"`, `"ja"`). Falls back to the constructor value. */
33
+ language?: string
34
+ /**
35
+ * Whisper task. Default `transcribe` keeps the source language.
36
+ * `translate` turns speech into English. Never the default here.
37
+ */
38
+ task?: 'transcribe' | 'translate'
39
+ }
40
+
41
+ export type STTAudioInput = Blob | File | string | Float32Array
42
+
43
+ type RawChunk = { text: string; timestamp: [number, number | null] }
44
+ type RawResult = { text: string; chunks?: RawChunk[] }
45
+
46
+ type ASRPipeline = {
47
+ (audio: Float32Array | string, options?: Record<string, unknown>): Promise<RawResult | RawResult[]>
48
+ tokenizer?: unknown
49
+ dispose?: () => Promise<void>
50
+ }
51
+
52
+ /**
53
+ * Speech-to-text that runs entirely in the browser (Whisper via Transformers.js).
54
+ *
55
+ * @example
56
+ * ```ts
57
+ * import { SpeechToText } from 'runonweb/stt'
58
+ *
59
+ * const stt = new SpeechToText()
60
+ * await stt.load()
61
+ * const { text } = await stt.transcribe(audioBlob)
62
+ * console.log(text)
63
+ * ```
64
+ */
65
+ export class SpeechToText {
66
+ #model: string
67
+ #device: Device
68
+ #language?: string
69
+ #onProgress?: ProgressCallback
70
+ #pipe: ASRPipeline | null = null
71
+ #loading: Promise<void> | null = null
72
+ #resolvedDevice: 'webgpu' | 'wasm' | null = null
73
+
74
+ constructor(options: STTOptions = {}) {
75
+ this.#model = options.model ?? DEFAULT_MODEL
76
+ this.#device = options.device ?? 'auto'
77
+ this.#language = options.language
78
+ this.#onProgress = options.onProgress
79
+ }
80
+
81
+ /** Device actually used after `load()`. */
82
+ get device(): 'webgpu' | 'wasm' | null {
83
+ return this.#resolvedDevice
84
+ }
85
+
86
+ /** Download and initialize the model. Safe to call multiple times. */
87
+ async load(): Promise<void> {
88
+ if (this.#pipe) return
89
+ if (this.#loading) return this.#loading
90
+
91
+ this.#loading = (async () => {
92
+ const { pipe, device } = await loadPipeline({
93
+ task: 'automatic-speech-recognition',
94
+ model: this.#model,
95
+ device: this.#device,
96
+ onProgress: this.#onProgress,
97
+ })
98
+ this.#pipe = pipe as unknown as ASRPipeline
99
+ this.#resolvedDevice = device
100
+ })()
101
+
102
+ try {
103
+ await this.#loading
104
+ } finally {
105
+ this.#loading = null
106
+ }
107
+ }
108
+
109
+ /**
110
+ * Transcribe audio. Accepts a Blob/File, a URL string, or raw Float32Array samples (16 kHz mono).
111
+ * Pass `onPartial` to receive words as they are generated.
112
+ */
113
+ async transcribe(audio: STTAudioInput, options?: TranscribeOptions): Promise<STTResult> {
114
+ await this.load()
115
+ if (!this.#pipe) throw new Error('SpeechToText model failed to load')
116
+
117
+ this.#onProgress?.({ status: 'transcribing' })
118
+
119
+ const input = await prepareAudio(audio)
120
+ const duration = input instanceof Float32Array ? input.length / SAMPLE_RATE : 0
121
+ const onPartial = options?.onPartial
122
+ let streamed = ''
123
+
124
+ const base: Record<string, unknown> = {
125
+ chunk_length_s: 30,
126
+ stride_length_s: 5,
127
+ // Always transcribe unless asked otherwise. Multilingual Whisper will
128
+ // otherwise slip into "translate to English" when language is unknown.
129
+ task: options?.task ?? 'transcribe',
130
+ // Word-timestamp mode disables timestamp tokens. Without this, decode
131
+ // throws "Whisper did not predict an ending timestamp" and the retry
132
+ // can come back empty after the streamer already showed text.
133
+ force_full_sequences: false,
134
+ }
135
+ const language = options?.language ?? this.#language
136
+ if (language) base.language = language
137
+
138
+ const run = async (returnTimestamps: true | 'word', stream: boolean) => {
139
+ const streamer =
140
+ stream && onPartial ? await createStreamer(this.#pipe!, (text) => {
141
+ streamed = text
142
+ onPartial(text)
143
+ }) : undefined
144
+ return this.#pipe!(input, {
145
+ ...base,
146
+ return_timestamps: returnTimestamps,
147
+ ...(streamer ? { streamer } : {}),
148
+ })
149
+ }
150
+
151
+ let raw: RawResult | RawResult[]
152
+ try {
153
+ // Segment timestamps + streamer is the reliable path. Word timestamps
154
+ // need a different generate() return shape and often fail once a streamer
155
+ // is attached, wiping the final text after a good partial stream.
156
+ raw = await run(true, true)
157
+ } catch {
158
+ raw = await run(true, false)
159
+ }
160
+
161
+ const parsed = normalizeRaw(raw)
162
+ const text = parsed.text || streamed
163
+ const chunks = explodeWords(parsed.chunks.length ? parsed.chunks : wordsFromText(text, duration), duration)
164
+
165
+ this.#onProgress?.({ status: 'done' })
166
+ return { text, chunks }
167
+ }
168
+
169
+ /** Release model resources. */
170
+ dispose(): void {
171
+ const pipe = this.#pipe
172
+ this.#pipe = null
173
+ this.#resolvedDevice = null
174
+ void pipe?.dispose?.()
175
+ }
176
+ }
177
+
178
+ /**
179
+ * One-shot helper: load model, transcribe, dispose.
180
+ */
181
+ export async function transcribe(
182
+ audio: STTAudioInput,
183
+ options?: STTOptions & TranscribeOptions
184
+ ): Promise<STTResult> {
185
+ const stt = new SpeechToText(options)
186
+ try {
187
+ return await stt.transcribe(audio, options)
188
+ } finally {
189
+ stt.dispose()
190
+ }
191
+ }
192
+
193
+ async function createStreamer(pipe: ASRPipeline, onPartial: (text: string) => void) {
194
+ const { WhisperTextStreamer } = await import('@huggingface/transformers')
195
+ let acc = ''
196
+ return new WhisperTextStreamer(pipe.tokenizer as never, {
197
+ skip_prompt: true,
198
+ skip_special_tokens: true,
199
+ callback_function: (piece: string) => {
200
+ acc += piece
201
+ const next = acc.replace(/\s+/g, ' ').trim()
202
+ if (next) onPartial(next)
203
+ },
204
+ })
205
+ }
206
+
207
+ function normalizeRaw(raw: RawResult | RawResult[]): { text: string; chunks: STTChunk[] } {
208
+ const items = Array.isArray(raw) ? raw : [raw]
209
+ const text = items
210
+ .map((item) => item.text ?? '')
211
+ .join(' ')
212
+ .replace(/\s+/g, ' ')
213
+ .trim()
214
+ return { text, chunks: items.flatMap((item) => parseChunks(item.chunks)) }
215
+ }
216
+
217
+ function parseChunks(raw?: RawChunk[]): STTChunk[] {
218
+ if (!raw?.length) return []
219
+ const chunks: STTChunk[] = []
220
+ for (const item of raw) {
221
+ const text = (item.text ?? '').replace(/\s+/g, ' ')
222
+ if (!text.trim()) continue
223
+ const start = item.timestamp?.[0] ?? chunks.at(-1)?.end ?? 0
224
+ const end = item.timestamp?.[1] ?? start
225
+ chunks.push({ text, start, end: end < start ? start : end })
226
+ }
227
+ return chunks
228
+ }
229
+
230
+ function wordsFromText(text: string, duration: number): STTChunk[] {
231
+ const parts = text.trim().split(/\s+/).filter(Boolean)
232
+ if (!parts.length) return []
233
+ const span = Math.max(0.04, (duration || parts.length * 0.35) / parts.length)
234
+ return parts.map((word, i) => ({
235
+ text: word,
236
+ start: i * span,
237
+ end: (i + 1) * span,
238
+ }))
239
+ }
240
+
241
+ /** Split segment chunks into words and align timestamps to the audio duration. */
242
+ function explodeWords(raw: STTChunk[], duration: number): STTChunk[] {
243
+ const words: STTChunk[] = []
244
+ for (const chunk of raw) {
245
+ const parts = chunk.text.trim().split(/\s+/).filter(Boolean)
246
+ if (!parts.length) continue
247
+ if (parts.length === 1) {
248
+ words.push({ text: parts[0]!, start: chunk.start, end: Math.max(chunk.end, chunk.start) })
249
+ continue
250
+ }
251
+ const start = chunk.start
252
+ const end = Math.max(chunk.end, chunk.start)
253
+ const span = Math.max(0.04, (end - start || parts.length * 0.35) / parts.length)
254
+ parts.forEach((word, i) => {
255
+ words.push({
256
+ text: word,
257
+ start: start + i * span,
258
+ end: start + (i + 1) * span,
259
+ })
260
+ })
261
+ }
262
+ return alignToDuration(words, duration)
263
+ }
264
+
265
+ function alignToDuration(chunks: STTChunk[], duration: number): STTChunk[] {
266
+ if (!chunks.length) return chunks
267
+ const maxEnd = Math.max(...chunks.map((c) => Math.max(c.end, c.start)))
268
+ if (maxEnd <= 0.05 && duration > 0) return wordsFromText(chunks.map((c) => c.text).join(' '), duration)
269
+ if (duration > 0 && (maxEnd > duration * 1.2 || maxEnd < duration * 0.55)) {
270
+ const scale = duration / maxEnd
271
+ return chunks.map((c) => ({ ...c, start: c.start * scale, end: Math.max(c.end, c.start) * scale }))
272
+ }
273
+ return chunks
274
+ }
275
+
276
+ export type { ProgressInfo, Device }
277
+
278
+ async function prepareAudio(audio: STTAudioInput): Promise<Float32Array | string> {
279
+ if (typeof audio === 'string') return audio
280
+ if (audio instanceof Float32Array) return audio
281
+
282
+ const arrayBuffer = await audio.arrayBuffer()
283
+ const AudioCtx =
284
+ globalThis.AudioContext ??
285
+ (globalThis as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext
286
+ const ctx = new AudioCtx()
287
+ try {
288
+ const decoded = await ctx.decodeAudioData(arrayBuffer.slice(0))
289
+ const mono = mixMono(decoded)
290
+ return resample(mono, decoded.sampleRate, SAMPLE_RATE)
291
+ } finally {
292
+ await ctx.close()
293
+ }
294
+ }
295
+
296
+ function mixMono(buffer: AudioBuffer): Float32Array {
297
+ if (buffer.numberOfChannels === 1) return new Float32Array(buffer.getChannelData(0))
298
+ const len = buffer.length
299
+ const out = new Float32Array(len)
300
+ const scale = buffer.numberOfChannels === 2 ? Math.SQRT1_2 : 1 / buffer.numberOfChannels
301
+ for (let c = 0; c < buffer.numberOfChannels; c++) {
302
+ const channel = buffer.getChannelData(c)
303
+ for (let i = 0; i < len; i++) out[i] += channel[i]! * scale
304
+ }
305
+ return out
306
+ }
307
+
308
+ function resample(input: Float32Array, from: number, to: number): Float32Array {
309
+ if (from === to) return input
310
+ const ratio = from / to
311
+ const out = new Float32Array(Math.round(input.length / ratio))
312
+ for (let i = 0; i < out.length; i++) {
313
+ const x = i * ratio
314
+ const i0 = Math.min(Math.floor(x), input.length - 1)
315
+ const i1 = Math.min(i0 + 1, input.length - 1)
316
+ const t = x - i0
317
+ out[i] = input[i0]! * (1 - t) + input[i1]! * t
318
+ }
319
+ return out
320
+ }