runonweb 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,165 @@
1
+ # runonweb
2
+
3
+ Free browser ML modules. Models run **locally in the browser** (WebGPU preferred, WASM fallback). No API keys, no servers. MIT.
4
+
5
+ ```bash
6
+ pnpm add runonweb
7
+ npm install runonweb
8
+ bun add runonweb
9
+ yarn add runonweb
10
+ ```
11
+
12
+ Every module follows the same shape:
13
+
14
+ ```ts
15
+ const m = new Module({ onProgress })
16
+ await m.load() // downloads + caches weights, safe to call twice
17
+ await m.<task>(input) // the actual work
18
+ m.dispose() // free memory
19
+ ```
20
+
21
+ Each module also exports a one-shot function (`transcribe`, `removeBackground`, `caption`, …) that loads, runs and disposes.
22
+
23
+ ## Modules
24
+
25
+ ### Speech-to-text: `runonweb/stt`
26
+
27
+ ```ts
28
+ import { SpeechToText } from 'runonweb/stt'
29
+
30
+ const stt = new SpeechToText()
31
+ await stt.load()
32
+ const { text } = await stt.transcribe(audioBlob, {
33
+ onPartial: (t) => console.log(t), // words as Whisper emits them
34
+ })
35
+ ```
36
+
37
+ Base: Whisper tiny.en (OpenAI, Apache-2.0). Pass `model` + `language` for multilingual Whisper variants.
38
+
39
+ ### Background removal: `runonweb/remove-bg`
40
+
41
+ ```ts
42
+ import { RemoveBackground } from 'runonweb/remove-bg'
43
+
44
+ const remover = new RemoveBackground()
45
+ const png = await remover.remove(imageFile) // PNG Blob with alpha
46
+ ```
47
+
48
+ Base: BEN2 (Prama LLC, MIT, ~219 MB fp16). Hair, objects and hard edges. WASM only until onnxruntime-web ships the LayerNorm shader fix (onnxruntime#32629).
49
+
50
+ ### Image captioning: `runonweb/caption`
51
+
52
+ ```ts
53
+ import { ImageCaptioner } from 'runonweb/caption'
54
+
55
+ const captioner = new ImageCaptioner({ language: 'es', detail: 'short' })
56
+ const { text } = await captioner.caption(imageFile)
57
+ ```
58
+
59
+ Base: LFM2.5-VL-450M (Liquid AI, ~316 MB on WebGPU). Captions in `en`, `es`, `pt`, `fr`, `de`, `ar`, `zh`, `ja`, `ko`; `detail` is `short` | `detailed` | `more`; `onPartial` streams.
60
+
61
+ > **License:** LFM Open License v1.0. Free for individuals and companies under USD 10M annual revenue; larger companies need a commercial license from Liquid AI. Not OSI-approved. See <https://huggingface.co/LiquidAI/LFM2.5-VL-450M/blob/main/LICENSE>.
62
+
63
+ ### Depth estimation: `runonweb/depth`
64
+
65
+ ```ts
66
+ import { DepthEstimator } from 'runonweb/depth'
67
+
68
+ const { depth, width, height } = await new DepthEstimator().estimate(imageFile)
69
+ // depth: grayscale PNG Blob, brighter = closer
70
+ ```
71
+
72
+ Base: Depth Anything V2 Small (Apache-2.0).
73
+
74
+ ### Object detection: `runonweb/detect`
75
+
76
+ ```ts
77
+ import { ObjectDetector } from 'runonweb/detect'
78
+
79
+ const objects = await new ObjectDetector({ threshold: 0.5 }).detect(imageFile)
80
+ // [{ label: 'cat', score: 0.98, box: { xmin, ymin, xmax, ymax } }]
81
+ ```
82
+
83
+ Base: DETR ResNet-50 (Meta, Apache-2.0). Boxes in pixel coordinates.
84
+
85
+ ### OCR: `runonweb/ocr`
86
+
87
+ ```ts
88
+ import { OCR } from 'runonweb/ocr'
89
+
90
+ const ocr = new OCR() // size: 'small', best size/quality
91
+ // const ocr = new OCR({ size: 'tiny' }) // ~6 MB
92
+ // const ocr = new OCR({ size: 'medium' }) // ~139 MB
93
+ const { text, lines } = await ocr.read(imageFile)
94
+ ```
95
+
96
+ Base: PP-OCRv6 (PaddlePaddle, Apache-2.0). `tiny` / `small` / `medium`. Default is `small` (~31 MB).
97
+
98
+ ### Text embeddings: `runonweb/embed`
99
+
100
+ ```ts
101
+ import { TextEmbedder, cosineSimilarity } from 'runonweb/embed'
102
+
103
+ const { embeddings, dimensions } = await new TextEmbedder().embed(['a', 'b'])
104
+ cosineSimilarity(embeddings[0], embeddings[1])
105
+ ```
106
+
107
+ Base: all-MiniLM-L6-v2 (Apache-2.0). 384 dims, normalized.
108
+
109
+ ### Translation: `runonweb/translate`
110
+
111
+ ```ts
112
+ import { Translator, PAIRS, hasPair } from 'runonweb/translate'
113
+
114
+ const t = new Translator({ from: 'en', to: 'es' }) // Firefox Translations en-es, ~37 MB
115
+ const { text } = await t.translate('Hello world')
116
+
117
+ await t.translate('Bonjour', { from: 'fr', to: 'en' }) // same instance, downloaded on demand
118
+ await t.translate('<b>Hello</b> world', { html: true }) // keeps the markup
119
+ hasPair('es', 'fr') // true, pivots through English
120
+
121
+ // one Transformers.js model for 100 languages (MIT, ~630 MB)
122
+ const multi = new Translator({ model: 'Xenova/m2m100_418M', from: 'fr', to: 'en' })
123
+ ```
124
+
125
+ Base: Mozilla's [Firefox Translations](https://github.com/mozilla/translations) models (MPL-2.0), the same Marian NMT students Firefox ships. 17–44 MB per pair (int8), 106 direct pairs between English and 58 languages; other pairs pivot through English. They run in a Web Worker through the Bergamot WASM runtime (no COOP/COEP headers needed) and are cached in Cache Storage. `PAIRS` lists every pair with architecture, size and COMET score.
126
+
127
+ Weights come from the runonweb mirror on the Hugging Face Hub by default. To self-host, run `node scripts/translate-models.mjs fetch en-es es-en` (or `--all`) and pass `modelPath` pointing at wherever you upload that folder; the script also copies the runtime so you can pass `runtimePath: '<modelPath>/runtime/'` instead of loading it from jsDelivr. The registry is regenerated from Mozilla's model list with `node scripts/translate-models.mjs registry`.
128
+
129
+ ### Text-to-speech: `runonweb/tts`
130
+
131
+ ```ts
132
+ import { TextToSpeech } from 'runonweb/tts'
133
+
134
+ const tts = new TextToSpeech({ size: 'small', voice: 'af_heart' })
135
+ await tts.load()
136
+
137
+ for await (const chunk of tts.speakStream('Hello from the browser')) {
138
+ // chunk.audio: 24 kHz PCM, one sentence at a time
139
+ }
140
+
141
+ const wav = await tts.speakToBlob('Hola mundo', { voice: 'ef_dora' })
142
+
143
+ // Supertonic 2: one model for en · ko · es · pt · fr, 44.1 kHz
144
+ const multi = new TextToSpeech({ size: 'multi', voice: 'st_f1' })
145
+ const pt = await multi.speakToBlob('Olá do navegador', { language: 'pt' })
146
+ ```
147
+
148
+ Base: Kokoro 82M (hexgrad, Apache-2.0) by default. English, Spanish, French, on Transformers.js directly. Pass `size: 'multi'` for Supertonic 2 (Supertone, OpenRAIL-M, ~262 MB): English, Korean, Spanish, Portuguese, French with 10 shared voices `st_f1`…`st_m5` and a `language` option. Pass `size: 'tiny'` for KittenTTS nano (~28 MB, 8 English voices). Audio streams via `speakStream`. Kokoro Spanish/French download a local eSpeak-NG WASM (~18 MB) on first use.
149
+
150
+ ### Core helpers: `runonweb/core`
151
+
152
+ ```ts
153
+ import { isWebGPUAvailable, resolveDevice } from 'runonweb/core'
154
+
155
+ const ok = await isWebGPUAvailable()
156
+ const device = await resolveDevice() // 'webgpu' | 'wasm'
157
+ ```
158
+
159
+ ## Notes
160
+
161
+ - First run downloads model weights from Hugging Face; the browser caches them afterward.
162
+ - Each module picks its quantization per device (`fp16`/`fp32` on WebGPU, `q8` on WASM). `q8` on WebGPU is never used because it produces wrong results.
163
+ - Models that fail on WebGPU in ONNX Runtime Web (detect) are pinned to WASM through `supportedDevices` and ignore `device: 'webgpu'`. Translation runs on CPU (Bergamot WASM) by design. Kokoro and Supertonic TTS use WebGPU; KittenTTS (`size: 'tiny'`) is WASM-only.
164
+ - Nothing leaves the browser. Audio, images and text stay on the user's device.
165
+ - All default models are Apache-2.0 or MIT except `runonweb/caption` (LFM Open License v1.0, free under USD 10M revenue). Attribution to the original authors is listed above and on every demo page.
package/package.json ADDED
@@ -0,0 +1,126 @@
1
+ {
2
+ "name": "runonweb",
3
+ "version": "0.0.1",
4
+ "description": "Free browser ML modules: speech, vision, text. Runs locally with WebGPU or WASM.",
5
+ "type": "module",
6
+ "license": "MIT",
7
+ "exports": {
8
+ "./core": {
9
+ "types": "./src/core/index.ts",
10
+ "import": "./src/core/index.ts",
11
+ "default": "./src/core/index.ts"
12
+ },
13
+ "./stt": {
14
+ "types": "./src/stt/index.ts",
15
+ "import": "./src/stt/index.ts",
16
+ "default": "./src/stt/index.ts"
17
+ },
18
+ "./remove-bg": {
19
+ "types": "./src/remove-bg/index.ts",
20
+ "import": "./src/remove-bg/index.ts",
21
+ "default": "./src/remove-bg/index.ts"
22
+ },
23
+ "./caption": {
24
+ "types": "./src/caption/index.ts",
25
+ "import": "./src/caption/index.ts",
26
+ "default": "./src/caption/index.ts"
27
+ },
28
+ "./depth": {
29
+ "types": "./src/depth/index.ts",
30
+ "import": "./src/depth/index.ts",
31
+ "default": "./src/depth/index.ts"
32
+ },
33
+ "./detect": {
34
+ "types": "./src/detect/index.ts",
35
+ "import": "./src/detect/index.ts",
36
+ "default": "./src/detect/index.ts"
37
+ },
38
+ "./ocr": {
39
+ "types": "./src/ocr/index.ts",
40
+ "import": "./src/ocr/index.ts",
41
+ "default": "./src/ocr/index.ts"
42
+ },
43
+ "./ocr/sizes": {
44
+ "types": "./src/ocr/sizes.ts",
45
+ "import": "./src/ocr/sizes.ts",
46
+ "default": "./src/ocr/sizes.ts"
47
+ },
48
+ "./embed": {
49
+ "types": "./src/embed/index.ts",
50
+ "import": "./src/embed/index.ts",
51
+ "default": "./src/embed/index.ts"
52
+ },
53
+ "./translate": {
54
+ "types": "./src/translate/index.ts",
55
+ "import": "./src/translate/index.ts",
56
+ "default": "./src/translate/index.ts"
57
+ },
58
+ "./tts": {
59
+ "types": "./src/tts/index.ts",
60
+ "import": "./src/tts/index.ts",
61
+ "default": "./src/tts/index.ts"
62
+ },
63
+ "./tts/voices": {
64
+ "types": "./src/tts/voices.ts",
65
+ "import": "./src/tts/voices.ts",
66
+ "default": "./src/tts/voices.ts"
67
+ },
68
+ "./tts/sizes": {
69
+ "types": "./src/tts/sizes.ts",
70
+ "import": "./src/tts/sizes.ts",
71
+ "default": "./src/tts/sizes.ts"
72
+ },
73
+ "./clean": {
74
+ "types": "./src/clean/index.ts",
75
+ "import": "./src/clean/index.ts",
76
+ "default": "./src/clean/index.ts"
77
+ },
78
+ "./emoji": {
79
+ "types": "./src/emoji/index.ts",
80
+ "import": "./src/emoji/index.ts",
81
+ "default": "./src/emoji/index.ts"
82
+ },
83
+ "./image": {
84
+ "types": "./src/image/index.ts",
85
+ "import": "./src/image/index.ts",
86
+ "default": "./src/image/index.ts"
87
+ },
88
+ "./image/sizes": {
89
+ "types": "./src/image/sizes.ts",
90
+ "import": "./src/image/sizes.ts",
91
+ "default": "./src/image/sizes.ts"
92
+ }
93
+ },
94
+ "files": [
95
+ "src"
96
+ ],
97
+ "sideEffects": false,
98
+ "dependencies": {
99
+ "@huggingface/transformers": "^4.3.0",
100
+ "onnxruntime-web": "^1.30.0",
101
+ "paddleocr": "^1.2.0",
102
+ "phonemizer": "^1.2.1"
103
+ },
104
+ "keywords": [
105
+ "webgpu",
106
+ "wasm",
107
+ "onnx",
108
+ "transformers",
109
+ "browser",
110
+ "ml",
111
+ "ai",
112
+ "whisper",
113
+ "speech-to-text",
114
+ "background-removal",
115
+ "ocr",
116
+ "embeddings",
117
+ "translation",
118
+ "tts",
119
+ "emoji",
120
+ "transcript-cleanup",
121
+ "dictation",
122
+ "image-generation",
123
+ "text-to-image",
124
+ "bonsai"
125
+ ]
126
+ }
@@ -0,0 +1,310 @@
1
+ import type { Device, ProgressCallback, ProgressInfo, ResolvedDevice } from '../core/index.ts'
2
+ import { resolveDevice } from '../core/device.ts'
3
+ import { imageToPipelineInput } from '../core/pipeline.ts'
4
+ import { toProgressInfo } from '../core/progress.ts'
5
+
6
+ /**
7
+ * Default: LFM2.5-VL-450M (Liquid AI, Nov 2025). Multilingual vision-language model.
8
+ *
9
+ * License: LFM Open License v1.0. Free for individuals and for companies under
10
+ * USD 10M annual revenue. Larger companies need a commercial license from Liquid AI.
11
+ * https://huggingface.co/LiquidAI/LFM2.5-VL-450M/blob/main/LICENSE
12
+ */
13
+ const DEFAULT_MODEL = 'onnx-community/LFM2.5-VL-450M-ONNX'
14
+
15
+ /** Languages the default model was trained on. Any other value is passed to the prompt as-is. */
16
+ export const CAPTION_LANGUAGES = {
17
+ en: 'English',
18
+ es: 'Spanish',
19
+ pt: 'Portuguese',
20
+ fr: 'French',
21
+ de: 'German',
22
+ ar: 'Arabic',
23
+ zh: 'Chinese',
24
+ ja: 'Japanese',
25
+ ko: 'Korean',
26
+ } as const
27
+
28
+ export type CaptionLanguage = keyof typeof CAPTION_LANGUAGES
29
+
30
+ const PROMPTS = {
31
+ short: 'Describe this image in one short sentence, suitable as alt text. Do not start with "This image" or "The image".',
32
+ detailed: 'Describe this image in two or three sentences.',
33
+ more: 'Describe this image in detail: subjects, setting, colors, composition and any visible text.',
34
+ } as const
35
+
36
+ const MAX_TOKENS = {
37
+ short: 48,
38
+ detailed: 128,
39
+ more: 320,
40
+ } as const
41
+
42
+ export type CaptionDetail = keyof typeof PROMPTS
43
+
44
+ export type CaptionOptions = {
45
+ model?: string
46
+ device?: Device
47
+ /** Caption verbosity. `short` is alt-text style. */
48
+ detail?: CaptionDetail
49
+ /**
50
+ * Caption language. A code from `CAPTION_LANGUAGES` or a plain language name ("Catalan").
51
+ * Default `en`. The model was trained on the 9 listed languages; others are best-effort.
52
+ */
53
+ language?: CaptionLanguage | (string & {})
54
+ /** Replace the built-in instruction. `language` is still appended. */
55
+ prompt?: string
56
+ /** Max new tokens for generation. Defaults depend on `detail`. */
57
+ maxNewTokens?: number
58
+ onProgress?: ProgressCallback
59
+ }
60
+
61
+ export type CaptionRunOptions = Pick<CaptionOptions, 'detail' | 'language' | 'prompt' | 'maxNewTokens'> & {
62
+ /** Receives the caption as it is generated. */
63
+ onPartial?: (text: string) => void
64
+ }
65
+
66
+ export type CaptionImageInput = Blob | File | string | HTMLImageElement | HTMLCanvasElement | ImageData
67
+
68
+ export type CaptionResult = {
69
+ text: string
70
+ }
71
+
72
+ type Tensor = {
73
+ dims: number[]
74
+ slice: (...args: unknown[]) => Tensor
75
+ }
76
+
77
+ type VlmModel = {
78
+ generate: (inputs: Record<string, unknown>) => Promise<Tensor>
79
+ dispose?: () => Promise<void>
80
+ }
81
+
82
+ type VlmProcessor = {
83
+ (image: unknown, text: string, options?: Record<string, unknown>): Promise<Record<string, unknown>>
84
+ apply_chat_template: (messages: unknown, options?: Record<string, unknown>) => string
85
+ batch_decode: (ids: Tensor, options: { skip_special_tokens: boolean }) => string[]
86
+ tokenizer: unknown
87
+ image_processor: { do_image_splitting?: boolean }
88
+ }
89
+
90
+ type RawImageLike = {
91
+ width: number
92
+ height: number
93
+ }
94
+
95
+ async function supportsShaderF16(): Promise<boolean> {
96
+ try {
97
+ const gpu = (globalThis as { navigator?: { gpu?: GPU } }).navigator?.gpu
98
+ const adapter = await gpu?.requestAdapter()
99
+ return Boolean(adapter?.features.has('shader-f16'))
100
+ } catch {
101
+ return false
102
+ }
103
+ }
104
+
105
+ function dtypeFor(device: ResolvedDevice, fp16: boolean) {
106
+ if (device === 'webgpu') {
107
+ // Official LFM2.5-VL WebGPU recipe: fp16 vision + embeddings, q4f16 decoder.
108
+ return fp16
109
+ ? ({ embed_tokens: 'fp16', vision_encoder: 'fp16', decoder_model_merged: 'q4f16' } as const)
110
+ : ({ embed_tokens: 'fp32', vision_encoder: 'fp32', decoder_model_merged: 'q4' } as const)
111
+ }
112
+ // WASM: q8/q4 embeddings use GatherBlockQuantized, which ONNX Runtime Web lacks on WASM. fp16 works.
113
+ return { embed_tokens: 'fp16', vision_encoder: 'q8', decoder_model_merged: 'q4' } as const
114
+ }
115
+
116
+ async function toRawImage(image: CaptionImageInput): Promise<RawImageLike> {
117
+ const input = await imageToPipelineInput(image)
118
+ if (typeof input !== 'string') return input as unknown as RawImageLike
119
+ const { RawImage } = await import('@huggingface/transformers')
120
+ return RawImage.fromURL(input) as unknown as Promise<RawImageLike>
121
+ }
122
+
123
+ function languageName(language: string): string {
124
+ return (CAPTION_LANGUAGES as Record<string, string>)[language] ?? language
125
+ }
126
+
127
+ function buildPrompt(detail: CaptionDetail, language: string, custom?: string): string {
128
+ const base = custom ?? PROMPTS[detail]
129
+ return `${base} Answer in ${languageName(language)}.`
130
+ }
131
+
132
+ /**
133
+ * LFM2.5-VL ships a chat template that uses the `{% generation %}` tag, which the Jinja
134
+ * engine bundled with Transformers.js 4.2 does not parse yet. Try the template first (so a
135
+ * custom `model` keeps its own format) and fall back to the documented ChatML layout.
136
+ */
137
+ function formatChat(processor: VlmProcessor, prompt: string): string {
138
+ try {
139
+ const messages = [{ role: 'user', content: [{ type: 'image' }, { type: 'text', text: prompt }] }]
140
+ return processor.apply_chat_template(messages, { add_generation_prompt: true })
141
+ } catch {
142
+ return `<|startoftext|><|im_start|>user\n<image>${prompt}<|im_end|>\n<|im_start|>assistant\n`
143
+ }
144
+ }
145
+
146
+ function cleanCaption(text: string): string {
147
+ const t = text.trim().replace(/^["'“”]+|["'“”]+$/g, '').trim()
148
+ return t.charAt(0).toUpperCase() + t.slice(1)
149
+ }
150
+
151
+ /**
152
+ * Image captioning in the browser (LFM2.5-VL-450M, multilingual).
153
+ *
154
+ * @example
155
+ * ```ts
156
+ * import { ImageCaptioner } from 'runonweb/caption'
157
+ *
158
+ * const captioner = new ImageCaptioner({ language: 'es' })
159
+ * await captioner.load()
160
+ * const { text } = await captioner.caption(imageFile)
161
+ * ```
162
+ */
163
+ export class ImageCaptioner {
164
+ #modelId: string
165
+ #device: Device
166
+ #detail: CaptionDetail
167
+ #language: string
168
+ #prompt?: string
169
+ #maxNewTokens?: number
170
+ #onProgress?: ProgressCallback
171
+ #model: VlmModel | null = null
172
+ #processor: VlmProcessor | null = null
173
+ #loading: Promise<void> | null = null
174
+ #resolvedDevice: 'webgpu' | 'wasm' | null = null
175
+
176
+ constructor(options: CaptionOptions = {}) {
177
+ this.#modelId = options.model ?? DEFAULT_MODEL
178
+ this.#device = options.device ?? 'auto'
179
+ this.#detail = options.detail ?? 'short'
180
+ this.#language = options.language ?? 'en'
181
+ this.#prompt = options.prompt
182
+ this.#maxNewTokens = options.maxNewTokens
183
+ this.#onProgress = options.onProgress
184
+ }
185
+
186
+ get device(): 'webgpu' | 'wasm' | null {
187
+ return this.#resolvedDevice
188
+ }
189
+
190
+ async load(): Promise<void> {
191
+ if (this.#model) return
192
+ if (this.#loading) return this.#loading
193
+
194
+ this.#loading = (async () => {
195
+ const preferred = await resolveDevice(this.#device)
196
+ try {
197
+ await this.#loadOn(preferred)
198
+ } catch (err) {
199
+ if (preferred === 'wasm' || this.#device === 'wasm') throw err
200
+ await this.#loadOn('wasm')
201
+ }
202
+ })()
203
+
204
+ try {
205
+ await this.#loading
206
+ } finally {
207
+ this.#loading = null
208
+ }
209
+ }
210
+
211
+ async #loadOn(device: ResolvedDevice): Promise<void> {
212
+ this.#onProgress?.({ status: 'loading', progress: 0 })
213
+
214
+ const { AutoModelForImageTextToText, AutoProcessor, env } = await import('@huggingface/transformers')
215
+ env.allowLocalModels = false
216
+
217
+ const fp16 = device === 'webgpu' && (await supportsShaderF16())
218
+ const progress_callback = (data: Record<string, unknown>) => {
219
+ this.#onProgress?.(toProgressInfo(data))
220
+ }
221
+
222
+ const [model, processor] = await Promise.all([
223
+ AutoModelForImageTextToText.from_pretrained(this.#modelId, {
224
+ device,
225
+ dtype: dtypeFor(device, fp16),
226
+ progress_callback,
227
+ }),
228
+ AutoProcessor.from_pretrained(this.#modelId, { progress_callback }),
229
+ ])
230
+
231
+ const proc = processor as unknown as VlmProcessor
232
+ // Images are resized to fit 512×512 anyway; tiling only adds tokens and latency for captions.
233
+ if (proc.image_processor) proc.image_processor.do_image_splitting = false
234
+
235
+ this.#model = model as unknown as VlmModel
236
+ this.#processor = proc
237
+ this.#resolvedDevice = device
238
+
239
+ this.#onProgress?.({ status: 'ready', progress: 100 })
240
+ }
241
+
242
+ async caption(image: CaptionImageInput, options: CaptionRunOptions = {}): Promise<CaptionResult> {
243
+ await this.load()
244
+ const model = this.#model
245
+ const processor = this.#processor
246
+ if (!model || !processor) {
247
+ throw new Error('ImageCaptioner model failed to load')
248
+ }
249
+
250
+ this.#onProgress?.({ status: 'captioning' })
251
+
252
+ const detail = options.detail ?? this.#detail
253
+ const language = options.language ?? this.#language
254
+ const prompt = buildPrompt(detail, language, options.prompt ?? this.#prompt)
255
+ const maxNewTokens = options.maxNewTokens ?? this.#maxNewTokens ?? MAX_TOKENS[detail]
256
+
257
+ const chat = formatChat(processor, prompt)
258
+
259
+ const rawImage = await toRawImage(image)
260
+ const inputs = await processor(rawImage, chat, { add_special_tokens: false })
261
+
262
+ let streamer: unknown
263
+ if (options.onPartial) {
264
+ const { TextStreamer } = await import('@huggingface/transformers')
265
+ let partial = ''
266
+ streamer = new TextStreamer(processor.tokenizer as never, {
267
+ skip_prompt: true,
268
+ skip_special_tokens: true,
269
+ callback_function: (chunk: string) => {
270
+ partial += chunk
271
+ options.onPartial?.(cleanCaption(partial))
272
+ },
273
+ })
274
+ }
275
+
276
+ const output = await model.generate({
277
+ ...inputs,
278
+ max_new_tokens: maxNewTokens,
279
+ do_sample: false,
280
+ repetition_penalty: 1.05,
281
+ ...(streamer ? { streamer } : {}),
282
+ })
283
+
284
+ const promptLength = (inputs.input_ids as Tensor).dims.at(-1) ?? 0
285
+ const decoded = processor.batch_decode(output.slice(null, [promptLength, null]), { skip_special_tokens: true })
286
+ const text = cleanCaption(decoded[0] ?? '')
287
+
288
+ this.#onProgress?.({ status: 'done' })
289
+ return { text }
290
+ }
291
+
292
+ dispose(): void {
293
+ const model = this.#model
294
+ this.#model = null
295
+ this.#processor = null
296
+ this.#resolvedDevice = null
297
+ void model?.dispose?.()
298
+ }
299
+ }
300
+
301
+ export async function caption(image: CaptionImageInput, options?: CaptionOptions): Promise<CaptionResult> {
302
+ const captioner = new ImageCaptioner(options)
303
+ try {
304
+ return await captioner.caption(image)
305
+ } finally {
306
+ captioner.dispose()
307
+ }
308
+ }
309
+
310
+ export type { ProgressInfo, Device }