runonweb 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -0
- package/package.json +126 -0
- package/src/caption/index.ts +310 -0
- package/src/clean/index.ts +263 -0
- package/src/core/cache.ts +248 -0
- package/src/core/device.ts +34 -0
- package/src/core/index.ts +18 -0
- package/src/core/pipeline.ts +108 -0
- package/src/core/progress.ts +20 -0
- package/src/depth/index.ts +127 -0
- package/src/detect/index.ts +274 -0
- package/src/embed/index.ts +172 -0
- package/src/emoji/index.ts +132 -0
- package/src/image/engine.d.ts +57 -0
- package/src/image/engine.js +14090 -0
- package/src/image/index.ts +230 -0
- package/src/image/sizes.ts +28 -0
- package/src/ocr/index.ts +416 -0
- package/src/ocr/sizes.ts +12 -0
- package/src/remove-bg/index.ts +170 -0
- package/src/stt/index.ts +320 -0
- package/src/translate/bergamot.ts +280 -0
- package/src/translate/index.ts +240 -0
- package/src/translate/registry.ts +133 -0
- package/src/translate/worker.ts +181 -0
- package/src/tts/index.ts +220 -0
- package/src/tts/kitten.ts +280 -0
- package/src/tts/kokoro.ts +140 -0
- package/src/tts/phonemes.ts +150 -0
- package/src/tts/sizes.ts +30 -0
- package/src/tts/split.ts +31 -0
- package/src/tts/supertonic.ts +126 -0
- package/src/tts/types.ts +11 -0
- package/src/tts/voices.ts +131 -0
- package/src/tts/wav.ts +52 -0
package/README.md
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# runonweb
|
|
2
|
+
|
|
3
|
+
Free browser ML modules. Models run **locally in the browser** (WebGPU preferred, WASM fallback). No API keys, no servers. MIT.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pnpm add runonweb
|
|
7
|
+
npm install runonweb
|
|
8
|
+
bun add runonweb
|
|
9
|
+
yarn add runonweb
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
Every module follows the same shape:
|
|
13
|
+
|
|
14
|
+
```ts
|
|
15
|
+
const m = new Module({ onProgress })
|
|
16
|
+
await m.load() // downloads + caches weights, safe to call twice
|
|
17
|
+
await m.<task>(input) // the actual work
|
|
18
|
+
m.dispose() // free memory
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
Each module also exports a one-shot function (`transcribe`, `removeBackground`, `caption`, …) that loads, runs and disposes.
|
|
22
|
+
|
|
23
|
+
## Modules
|
|
24
|
+
|
|
25
|
+
### Speech-to-text: `runonweb/stt`
|
|
26
|
+
|
|
27
|
+
```ts
|
|
28
|
+
import { SpeechToText } from 'runonweb/stt'
|
|
29
|
+
|
|
30
|
+
const stt = new SpeechToText()
|
|
31
|
+
await stt.load()
|
|
32
|
+
const { text } = await stt.transcribe(audioBlob, {
|
|
33
|
+
onPartial: (t) => console.log(t), // words as Whisper emits them
|
|
34
|
+
})
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Base: Whisper tiny.en (OpenAI, Apache-2.0). Pass `model` + `language` for multilingual Whisper variants.
|
|
38
|
+
|
|
39
|
+
### Background removal: `runonweb/remove-bg`
|
|
40
|
+
|
|
41
|
+
```ts
|
|
42
|
+
import { RemoveBackground } from 'runonweb/remove-bg'
|
|
43
|
+
|
|
44
|
+
const remover = new RemoveBackground()
|
|
45
|
+
const png = await remover.remove(imageFile) // PNG Blob with alpha
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Base: BEN2 (Prama LLC, MIT, ~219 MB fp16). Hair, objects and hard edges. WASM only until onnxruntime-web ships the LayerNorm shader fix (onnxruntime#32629).
|
|
49
|
+
|
|
50
|
+
### Image captioning: `runonweb/caption`
|
|
51
|
+
|
|
52
|
+
```ts
|
|
53
|
+
import { ImageCaptioner } from 'runonweb/caption'
|
|
54
|
+
|
|
55
|
+
const captioner = new ImageCaptioner({ language: 'es', detail: 'short' })
|
|
56
|
+
const { text } = await captioner.caption(imageFile)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Base: LFM2.5-VL-450M (Liquid AI, ~316 MB on WebGPU). Captions in `en`, `es`, `pt`, `fr`, `de`, `ar`, `zh`, `ja`, `ko`; `detail` is `short` | `detailed` | `more`; `onPartial` streams.
|
|
60
|
+
|
|
61
|
+
> **License:** LFM Open License v1.0. Free for individuals and companies under USD 10M annual revenue; larger companies need a commercial license from Liquid AI. Not OSI-approved. See <https://huggingface.co/LiquidAI/LFM2.5-VL-450M/blob/main/LICENSE>.
|
|
62
|
+
|
|
63
|
+
### Depth estimation: `runonweb/depth`
|
|
64
|
+
|
|
65
|
+
```ts
|
|
66
|
+
import { DepthEstimator } from 'runonweb/depth'
|
|
67
|
+
|
|
68
|
+
const { depth, width, height } = await new DepthEstimator().estimate(imageFile)
|
|
69
|
+
// depth: grayscale PNG Blob, brighter = closer
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Base: Depth Anything V2 Small (Apache-2.0).
|
|
73
|
+
|
|
74
|
+
### Object detection: `runonweb/detect`
|
|
75
|
+
|
|
76
|
+
```ts
|
|
77
|
+
import { ObjectDetector } from 'runonweb/detect'
|
|
78
|
+
|
|
79
|
+
const objects = await new ObjectDetector({ threshold: 0.5 }).detect(imageFile)
|
|
80
|
+
// [{ label: 'cat', score: 0.98, box: { xmin, ymin, xmax, ymax } }]
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Base: DETR ResNet-50 (Meta, Apache-2.0). Boxes in pixel coordinates.
|
|
84
|
+
|
|
85
|
+
### OCR: `runonweb/ocr`
|
|
86
|
+
|
|
87
|
+
```ts
|
|
88
|
+
import { OCR } from 'runonweb/ocr'
|
|
89
|
+
|
|
90
|
+
const ocr = new OCR() // size: 'small', best size/quality
|
|
91
|
+
// const ocr = new OCR({ size: 'tiny' }) // ~6 MB
|
|
92
|
+
// const ocr = new OCR({ size: 'medium' }) // ~139 MB
|
|
93
|
+
const { text, lines } = await ocr.read(imageFile)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Base: PP-OCRv6 (PaddlePaddle, Apache-2.0). `tiny` / `small` / `medium`. Default is `small` (~31 MB).
|
|
97
|
+
|
|
98
|
+
### Text embeddings: `runonweb/embed`
|
|
99
|
+
|
|
100
|
+
```ts
|
|
101
|
+
import { TextEmbedder, cosineSimilarity } from 'runonweb/embed'
|
|
102
|
+
|
|
103
|
+
const { embeddings, dimensions } = await new TextEmbedder().embed(['a', 'b'])
|
|
104
|
+
cosineSimilarity(embeddings[0], embeddings[1])
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Base: all-MiniLM-L6-v2 (Apache-2.0). 384 dims, normalized.
|
|
108
|
+
|
|
109
|
+
### Translation: `runonweb/translate`
|
|
110
|
+
|
|
111
|
+
```ts
|
|
112
|
+
import { Translator, PAIRS, hasPair } from 'runonweb/translate'
|
|
113
|
+
|
|
114
|
+
const t = new Translator({ from: 'en', to: 'es' }) // Firefox Translations en-es, ~37 MB
|
|
115
|
+
const { text } = await t.translate('Hello world')
|
|
116
|
+
|
|
117
|
+
await t.translate('Bonjour', { from: 'fr', to: 'en' }) // same instance, downloaded on demand
|
|
118
|
+
await t.translate('<b>Hello</b> world', { html: true }) // keeps the markup
|
|
119
|
+
hasPair('es', 'fr') // true, pivots through English
|
|
120
|
+
|
|
121
|
+
// one Transformers.js model for 100 languages (MIT, ~630 MB)
|
|
122
|
+
const multi = new Translator({ model: 'Xenova/m2m100_418M', from: 'fr', to: 'en' })
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Base: Mozilla's [Firefox Translations](https://github.com/mozilla/translations) models (MPL-2.0), the same Marian NMT students Firefox ships. 17–44 MB per pair (int8), 106 direct pairs between English and 58 languages; other pairs pivot through English. They run in a Web Worker through the Bergamot WASM runtime (no COOP/COEP headers needed) and are cached in Cache Storage. `PAIRS` lists every pair with architecture, size and COMET score.
|
|
126
|
+
|
|
127
|
+
Weights come from the runonweb mirror on the Hugging Face Hub by default. To self-host, run `node scripts/translate-models.mjs fetch en-es es-en` (or `--all`) and pass `modelPath` pointing at wherever you upload that folder; the script also copies the runtime so you can pass `runtimePath: '<modelPath>/runtime/'` instead of loading it from jsDelivr. The registry is regenerated from Mozilla's model list with `node scripts/translate-models.mjs registry`.
|
|
128
|
+
|
|
129
|
+
### Text-to-speech: `runonweb/tts`
|
|
130
|
+
|
|
131
|
+
```ts
|
|
132
|
+
import { TextToSpeech } from 'runonweb/tts'
|
|
133
|
+
|
|
134
|
+
const tts = new TextToSpeech({ size: 'small', voice: 'af_heart' })
|
|
135
|
+
await tts.load()
|
|
136
|
+
|
|
137
|
+
for await (const chunk of tts.speakStream('Hello from the browser')) {
|
|
138
|
+
// chunk.audio: 24 kHz PCM, one sentence at a time
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const wav = await tts.speakToBlob('Hola mundo', { voice: 'ef_dora' })
|
|
142
|
+
|
|
143
|
+
// Supertonic 2: one model for en · ko · es · pt · fr, 44.1 kHz
|
|
144
|
+
const multi = new TextToSpeech({ size: 'multi', voice: 'st_f1' })
|
|
145
|
+
const pt = await multi.speakToBlob('Olá do navegador', { language: 'pt' })
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Base: Kokoro 82M (hexgrad, Apache-2.0) by default. English, Spanish, French, on Transformers.js directly. Pass `size: 'multi'` for Supertonic 2 (Supertone, OpenRAIL-M, ~262 MB): English, Korean, Spanish, Portuguese, French with 10 shared voices `st_f1`…`st_m5` and a `language` option. Pass `size: 'tiny'` for KittenTTS nano (~28 MB, 8 English voices). Audio streams via `speakStream`. Kokoro Spanish/French download a local eSpeak-NG WASM (~18 MB) on first use.
|
|
149
|
+
|
|
150
|
+
### Core helpers: `runonweb/core`
|
|
151
|
+
|
|
152
|
+
```ts
|
|
153
|
+
import { isWebGPUAvailable, resolveDevice } from 'runonweb/core'
|
|
154
|
+
|
|
155
|
+
const ok = await isWebGPUAvailable()
|
|
156
|
+
const device = await resolveDevice() // 'webgpu' | 'wasm'
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
## Notes
|
|
160
|
+
|
|
161
|
+
- First run downloads model weights from Hugging Face; the browser caches them afterward.
|
|
162
|
+
- Each module picks its quantization per device (`fp16`/`fp32` on WebGPU, `q8` on WASM). `q8` on WebGPU is never used because it produces wrong results.
|
|
163
|
+
- Models that fail on WebGPU in ONNX Runtime Web (detect) are pinned to WASM through `supportedDevices` and ignore `device: 'webgpu'`. Translation runs on CPU (Bergamot WASM) by design. Kokoro and Supertonic TTS use WebGPU; KittenTTS (`size: 'tiny'`) is WASM-only.
|
|
164
|
+
- Nothing leaves the browser. Audio, images and text stay on the user's device.
|
|
165
|
+
- All default models are Apache-2.0 or MIT except `runonweb/caption` (LFM Open License v1.0, free under USD 10M revenue). Attribution to the original authors is listed above and on every demo page.
|
package/package.json
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "runonweb",
|
|
3
|
+
"version": "0.0.1",
|
|
4
|
+
"description": "Free browser ML modules: speech, vision, text. Runs locally with WebGPU or WASM.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"exports": {
|
|
8
|
+
"./core": {
|
|
9
|
+
"types": "./src/core/index.ts",
|
|
10
|
+
"import": "./src/core/index.ts",
|
|
11
|
+
"default": "./src/core/index.ts"
|
|
12
|
+
},
|
|
13
|
+
"./stt": {
|
|
14
|
+
"types": "./src/stt/index.ts",
|
|
15
|
+
"import": "./src/stt/index.ts",
|
|
16
|
+
"default": "./src/stt/index.ts"
|
|
17
|
+
},
|
|
18
|
+
"./remove-bg": {
|
|
19
|
+
"types": "./src/remove-bg/index.ts",
|
|
20
|
+
"import": "./src/remove-bg/index.ts",
|
|
21
|
+
"default": "./src/remove-bg/index.ts"
|
|
22
|
+
},
|
|
23
|
+
"./caption": {
|
|
24
|
+
"types": "./src/caption/index.ts",
|
|
25
|
+
"import": "./src/caption/index.ts",
|
|
26
|
+
"default": "./src/caption/index.ts"
|
|
27
|
+
},
|
|
28
|
+
"./depth": {
|
|
29
|
+
"types": "./src/depth/index.ts",
|
|
30
|
+
"import": "./src/depth/index.ts",
|
|
31
|
+
"default": "./src/depth/index.ts"
|
|
32
|
+
},
|
|
33
|
+
"./detect": {
|
|
34
|
+
"types": "./src/detect/index.ts",
|
|
35
|
+
"import": "./src/detect/index.ts",
|
|
36
|
+
"default": "./src/detect/index.ts"
|
|
37
|
+
},
|
|
38
|
+
"./ocr": {
|
|
39
|
+
"types": "./src/ocr/index.ts",
|
|
40
|
+
"import": "./src/ocr/index.ts",
|
|
41
|
+
"default": "./src/ocr/index.ts"
|
|
42
|
+
},
|
|
43
|
+
"./ocr/sizes": {
|
|
44
|
+
"types": "./src/ocr/sizes.ts",
|
|
45
|
+
"import": "./src/ocr/sizes.ts",
|
|
46
|
+
"default": "./src/ocr/sizes.ts"
|
|
47
|
+
},
|
|
48
|
+
"./embed": {
|
|
49
|
+
"types": "./src/embed/index.ts",
|
|
50
|
+
"import": "./src/embed/index.ts",
|
|
51
|
+
"default": "./src/embed/index.ts"
|
|
52
|
+
},
|
|
53
|
+
"./translate": {
|
|
54
|
+
"types": "./src/translate/index.ts",
|
|
55
|
+
"import": "./src/translate/index.ts",
|
|
56
|
+
"default": "./src/translate/index.ts"
|
|
57
|
+
},
|
|
58
|
+
"./tts": {
|
|
59
|
+
"types": "./src/tts/index.ts",
|
|
60
|
+
"import": "./src/tts/index.ts",
|
|
61
|
+
"default": "./src/tts/index.ts"
|
|
62
|
+
},
|
|
63
|
+
"./tts/voices": {
|
|
64
|
+
"types": "./src/tts/voices.ts",
|
|
65
|
+
"import": "./src/tts/voices.ts",
|
|
66
|
+
"default": "./src/tts/voices.ts"
|
|
67
|
+
},
|
|
68
|
+
"./tts/sizes": {
|
|
69
|
+
"types": "./src/tts/sizes.ts",
|
|
70
|
+
"import": "./src/tts/sizes.ts",
|
|
71
|
+
"default": "./src/tts/sizes.ts"
|
|
72
|
+
},
|
|
73
|
+
"./clean": {
|
|
74
|
+
"types": "./src/clean/index.ts",
|
|
75
|
+
"import": "./src/clean/index.ts",
|
|
76
|
+
"default": "./src/clean/index.ts"
|
|
77
|
+
},
|
|
78
|
+
"./emoji": {
|
|
79
|
+
"types": "./src/emoji/index.ts",
|
|
80
|
+
"import": "./src/emoji/index.ts",
|
|
81
|
+
"default": "./src/emoji/index.ts"
|
|
82
|
+
},
|
|
83
|
+
"./image": {
|
|
84
|
+
"types": "./src/image/index.ts",
|
|
85
|
+
"import": "./src/image/index.ts",
|
|
86
|
+
"default": "./src/image/index.ts"
|
|
87
|
+
},
|
|
88
|
+
"./image/sizes": {
|
|
89
|
+
"types": "./src/image/sizes.ts",
|
|
90
|
+
"import": "./src/image/sizes.ts",
|
|
91
|
+
"default": "./src/image/sizes.ts"
|
|
92
|
+
}
|
|
93
|
+
},
|
|
94
|
+
"files": [
|
|
95
|
+
"src"
|
|
96
|
+
],
|
|
97
|
+
"sideEffects": false,
|
|
98
|
+
"dependencies": {
|
|
99
|
+
"@huggingface/transformers": "^4.3.0",
|
|
100
|
+
"onnxruntime-web": "^1.30.0",
|
|
101
|
+
"paddleocr": "^1.2.0",
|
|
102
|
+
"phonemizer": "^1.2.1"
|
|
103
|
+
},
|
|
104
|
+
"keywords": [
|
|
105
|
+
"webgpu",
|
|
106
|
+
"wasm",
|
|
107
|
+
"onnx",
|
|
108
|
+
"transformers",
|
|
109
|
+
"browser",
|
|
110
|
+
"ml",
|
|
111
|
+
"ai",
|
|
112
|
+
"whisper",
|
|
113
|
+
"speech-to-text",
|
|
114
|
+
"background-removal",
|
|
115
|
+
"ocr",
|
|
116
|
+
"embeddings",
|
|
117
|
+
"translation",
|
|
118
|
+
"tts",
|
|
119
|
+
"emoji",
|
|
120
|
+
"transcript-cleanup",
|
|
121
|
+
"dictation",
|
|
122
|
+
"image-generation",
|
|
123
|
+
"text-to-image",
|
|
124
|
+
"bonsai"
|
|
125
|
+
]
|
|
126
|
+
}
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
import type { Device, ProgressCallback, ProgressInfo, ResolvedDevice } from '../core/index.ts'
|
|
2
|
+
import { resolveDevice } from '../core/device.ts'
|
|
3
|
+
import { imageToPipelineInput } from '../core/pipeline.ts'
|
|
4
|
+
import { toProgressInfo } from '../core/progress.ts'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Default: LFM2.5-VL-450M (Liquid AI, Nov 2025). Multilingual vision-language model.
|
|
8
|
+
*
|
|
9
|
+
* License: LFM Open License v1.0. Free for individuals and for companies under
|
|
10
|
+
* USD 10M annual revenue. Larger companies need a commercial license from Liquid AI.
|
|
11
|
+
* https://huggingface.co/LiquidAI/LFM2.5-VL-450M/blob/main/LICENSE
|
|
12
|
+
*/
|
|
13
|
+
const DEFAULT_MODEL = 'onnx-community/LFM2.5-VL-450M-ONNX'
|
|
14
|
+
|
|
15
|
+
/** Languages the default model was trained on. Any other value is passed to the prompt as-is. */
|
|
16
|
+
export const CAPTION_LANGUAGES = {
|
|
17
|
+
en: 'English',
|
|
18
|
+
es: 'Spanish',
|
|
19
|
+
pt: 'Portuguese',
|
|
20
|
+
fr: 'French',
|
|
21
|
+
de: 'German',
|
|
22
|
+
ar: 'Arabic',
|
|
23
|
+
zh: 'Chinese',
|
|
24
|
+
ja: 'Japanese',
|
|
25
|
+
ko: 'Korean',
|
|
26
|
+
} as const
|
|
27
|
+
|
|
28
|
+
export type CaptionLanguage = keyof typeof CAPTION_LANGUAGES
|
|
29
|
+
|
|
30
|
+
const PROMPTS = {
|
|
31
|
+
short: 'Describe this image in one short sentence, suitable as alt text. Do not start with "This image" or "The image".',
|
|
32
|
+
detailed: 'Describe this image in two or three sentences.',
|
|
33
|
+
more: 'Describe this image in detail: subjects, setting, colors, composition and any visible text.',
|
|
34
|
+
} as const
|
|
35
|
+
|
|
36
|
+
const MAX_TOKENS = {
|
|
37
|
+
short: 48,
|
|
38
|
+
detailed: 128,
|
|
39
|
+
more: 320,
|
|
40
|
+
} as const
|
|
41
|
+
|
|
42
|
+
export type CaptionDetail = keyof typeof PROMPTS
|
|
43
|
+
|
|
44
|
+
export type CaptionOptions = {
|
|
45
|
+
model?: string
|
|
46
|
+
device?: Device
|
|
47
|
+
/** Caption verbosity. `short` is alt-text style. */
|
|
48
|
+
detail?: CaptionDetail
|
|
49
|
+
/**
|
|
50
|
+
* Caption language. A code from `CAPTION_LANGUAGES` or a plain language name ("Catalan").
|
|
51
|
+
* Default `en`. The model was trained on the 9 listed languages; others are best-effort.
|
|
52
|
+
*/
|
|
53
|
+
language?: CaptionLanguage | (string & {})
|
|
54
|
+
/** Replace the built-in instruction. `language` is still appended. */
|
|
55
|
+
prompt?: string
|
|
56
|
+
/** Max new tokens for generation. Defaults depend on `detail`. */
|
|
57
|
+
maxNewTokens?: number
|
|
58
|
+
onProgress?: ProgressCallback
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export type CaptionRunOptions = Pick<CaptionOptions, 'detail' | 'language' | 'prompt' | 'maxNewTokens'> & {
|
|
62
|
+
/** Receives the caption as it is generated. */
|
|
63
|
+
onPartial?: (text: string) => void
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export type CaptionImageInput = Blob | File | string | HTMLImageElement | HTMLCanvasElement | ImageData
|
|
67
|
+
|
|
68
|
+
export type CaptionResult = {
|
|
69
|
+
text: string
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
type Tensor = {
|
|
73
|
+
dims: number[]
|
|
74
|
+
slice: (...args: unknown[]) => Tensor
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
type VlmModel = {
|
|
78
|
+
generate: (inputs: Record<string, unknown>) => Promise<Tensor>
|
|
79
|
+
dispose?: () => Promise<void>
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
type VlmProcessor = {
|
|
83
|
+
(image: unknown, text: string, options?: Record<string, unknown>): Promise<Record<string, unknown>>
|
|
84
|
+
apply_chat_template: (messages: unknown, options?: Record<string, unknown>) => string
|
|
85
|
+
batch_decode: (ids: Tensor, options: { skip_special_tokens: boolean }) => string[]
|
|
86
|
+
tokenizer: unknown
|
|
87
|
+
image_processor: { do_image_splitting?: boolean }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
type RawImageLike = {
|
|
91
|
+
width: number
|
|
92
|
+
height: number
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
async function supportsShaderF16(): Promise<boolean> {
|
|
96
|
+
try {
|
|
97
|
+
const gpu = (globalThis as { navigator?: { gpu?: GPU } }).navigator?.gpu
|
|
98
|
+
const adapter = await gpu?.requestAdapter()
|
|
99
|
+
return Boolean(adapter?.features.has('shader-f16'))
|
|
100
|
+
} catch {
|
|
101
|
+
return false
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function dtypeFor(device: ResolvedDevice, fp16: boolean) {
|
|
106
|
+
if (device === 'webgpu') {
|
|
107
|
+
// Official LFM2.5-VL WebGPU recipe: fp16 vision + embeddings, q4f16 decoder.
|
|
108
|
+
return fp16
|
|
109
|
+
? ({ embed_tokens: 'fp16', vision_encoder: 'fp16', decoder_model_merged: 'q4f16' } as const)
|
|
110
|
+
: ({ embed_tokens: 'fp32', vision_encoder: 'fp32', decoder_model_merged: 'q4' } as const)
|
|
111
|
+
}
|
|
112
|
+
// WASM: q8/q4 embeddings use GatherBlockQuantized, which ONNX Runtime Web lacks on WASM. fp16 works.
|
|
113
|
+
return { embed_tokens: 'fp16', vision_encoder: 'q8', decoder_model_merged: 'q4' } as const
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
async function toRawImage(image: CaptionImageInput): Promise<RawImageLike> {
|
|
117
|
+
const input = await imageToPipelineInput(image)
|
|
118
|
+
if (typeof input !== 'string') return input as unknown as RawImageLike
|
|
119
|
+
const { RawImage } = await import('@huggingface/transformers')
|
|
120
|
+
return RawImage.fromURL(input) as unknown as Promise<RawImageLike>
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function languageName(language: string): string {
|
|
124
|
+
return (CAPTION_LANGUAGES as Record<string, string>)[language] ?? language
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function buildPrompt(detail: CaptionDetail, language: string, custom?: string): string {
|
|
128
|
+
const base = custom ?? PROMPTS[detail]
|
|
129
|
+
return `${base} Answer in ${languageName(language)}.`
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* LFM2.5-VL ships a chat template that uses the `{% generation %}` tag, which the Jinja
|
|
134
|
+
* engine bundled with Transformers.js 4.2 does not parse yet. Try the template first (so a
|
|
135
|
+
* custom `model` keeps its own format) and fall back to the documented ChatML layout.
|
|
136
|
+
*/
|
|
137
|
+
function formatChat(processor: VlmProcessor, prompt: string): string {
|
|
138
|
+
try {
|
|
139
|
+
const messages = [{ role: 'user', content: [{ type: 'image' }, { type: 'text', text: prompt }] }]
|
|
140
|
+
return processor.apply_chat_template(messages, { add_generation_prompt: true })
|
|
141
|
+
} catch {
|
|
142
|
+
return `<|startoftext|><|im_start|>user\n<image>${prompt}<|im_end|>\n<|im_start|>assistant\n`
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
function cleanCaption(text: string): string {
|
|
147
|
+
const t = text.trim().replace(/^["'“”]+|["'“”]+$/g, '').trim()
|
|
148
|
+
return t.charAt(0).toUpperCase() + t.slice(1)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Image captioning in the browser (LFM2.5-VL-450M, multilingual).
|
|
153
|
+
*
|
|
154
|
+
* @example
|
|
155
|
+
* ```ts
|
|
156
|
+
* import { ImageCaptioner } from 'runonweb/caption'
|
|
157
|
+
*
|
|
158
|
+
* const captioner = new ImageCaptioner({ language: 'es' })
|
|
159
|
+
* await captioner.load()
|
|
160
|
+
* const { text } = await captioner.caption(imageFile)
|
|
161
|
+
* ```
|
|
162
|
+
*/
|
|
163
|
+
export class ImageCaptioner {
|
|
164
|
+
#modelId: string
|
|
165
|
+
#device: Device
|
|
166
|
+
#detail: CaptionDetail
|
|
167
|
+
#language: string
|
|
168
|
+
#prompt?: string
|
|
169
|
+
#maxNewTokens?: number
|
|
170
|
+
#onProgress?: ProgressCallback
|
|
171
|
+
#model: VlmModel | null = null
|
|
172
|
+
#processor: VlmProcessor | null = null
|
|
173
|
+
#loading: Promise<void> | null = null
|
|
174
|
+
#resolvedDevice: 'webgpu' | 'wasm' | null = null
|
|
175
|
+
|
|
176
|
+
constructor(options: CaptionOptions = {}) {
|
|
177
|
+
this.#modelId = options.model ?? DEFAULT_MODEL
|
|
178
|
+
this.#device = options.device ?? 'auto'
|
|
179
|
+
this.#detail = options.detail ?? 'short'
|
|
180
|
+
this.#language = options.language ?? 'en'
|
|
181
|
+
this.#prompt = options.prompt
|
|
182
|
+
this.#maxNewTokens = options.maxNewTokens
|
|
183
|
+
this.#onProgress = options.onProgress
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
get device(): 'webgpu' | 'wasm' | null {
|
|
187
|
+
return this.#resolvedDevice
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
async load(): Promise<void> {
|
|
191
|
+
if (this.#model) return
|
|
192
|
+
if (this.#loading) return this.#loading
|
|
193
|
+
|
|
194
|
+
this.#loading = (async () => {
|
|
195
|
+
const preferred = await resolveDevice(this.#device)
|
|
196
|
+
try {
|
|
197
|
+
await this.#loadOn(preferred)
|
|
198
|
+
} catch (err) {
|
|
199
|
+
if (preferred === 'wasm' || this.#device === 'wasm') throw err
|
|
200
|
+
await this.#loadOn('wasm')
|
|
201
|
+
}
|
|
202
|
+
})()
|
|
203
|
+
|
|
204
|
+
try {
|
|
205
|
+
await this.#loading
|
|
206
|
+
} finally {
|
|
207
|
+
this.#loading = null
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
async #loadOn(device: ResolvedDevice): Promise<void> {
|
|
212
|
+
this.#onProgress?.({ status: 'loading', progress: 0 })
|
|
213
|
+
|
|
214
|
+
const { AutoModelForImageTextToText, AutoProcessor, env } = await import('@huggingface/transformers')
|
|
215
|
+
env.allowLocalModels = false
|
|
216
|
+
|
|
217
|
+
const fp16 = device === 'webgpu' && (await supportsShaderF16())
|
|
218
|
+
const progress_callback = (data: Record<string, unknown>) => {
|
|
219
|
+
this.#onProgress?.(toProgressInfo(data))
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
const [model, processor] = await Promise.all([
|
|
223
|
+
AutoModelForImageTextToText.from_pretrained(this.#modelId, {
|
|
224
|
+
device,
|
|
225
|
+
dtype: dtypeFor(device, fp16),
|
|
226
|
+
progress_callback,
|
|
227
|
+
}),
|
|
228
|
+
AutoProcessor.from_pretrained(this.#modelId, { progress_callback }),
|
|
229
|
+
])
|
|
230
|
+
|
|
231
|
+
const proc = processor as unknown as VlmProcessor
|
|
232
|
+
// Images are resized to fit 512×512 anyway; tiling only adds tokens and latency for captions.
|
|
233
|
+
if (proc.image_processor) proc.image_processor.do_image_splitting = false
|
|
234
|
+
|
|
235
|
+
this.#model = model as unknown as VlmModel
|
|
236
|
+
this.#processor = proc
|
|
237
|
+
this.#resolvedDevice = device
|
|
238
|
+
|
|
239
|
+
this.#onProgress?.({ status: 'ready', progress: 100 })
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
async caption(image: CaptionImageInput, options: CaptionRunOptions = {}): Promise<CaptionResult> {
|
|
243
|
+
await this.load()
|
|
244
|
+
const model = this.#model
|
|
245
|
+
const processor = this.#processor
|
|
246
|
+
if (!model || !processor) {
|
|
247
|
+
throw new Error('ImageCaptioner model failed to load')
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
this.#onProgress?.({ status: 'captioning' })
|
|
251
|
+
|
|
252
|
+
const detail = options.detail ?? this.#detail
|
|
253
|
+
const language = options.language ?? this.#language
|
|
254
|
+
const prompt = buildPrompt(detail, language, options.prompt ?? this.#prompt)
|
|
255
|
+
const maxNewTokens = options.maxNewTokens ?? this.#maxNewTokens ?? MAX_TOKENS[detail]
|
|
256
|
+
|
|
257
|
+
const chat = formatChat(processor, prompt)
|
|
258
|
+
|
|
259
|
+
const rawImage = await toRawImage(image)
|
|
260
|
+
const inputs = await processor(rawImage, chat, { add_special_tokens: false })
|
|
261
|
+
|
|
262
|
+
let streamer: unknown
|
|
263
|
+
if (options.onPartial) {
|
|
264
|
+
const { TextStreamer } = await import('@huggingface/transformers')
|
|
265
|
+
let partial = ''
|
|
266
|
+
streamer = new TextStreamer(processor.tokenizer as never, {
|
|
267
|
+
skip_prompt: true,
|
|
268
|
+
skip_special_tokens: true,
|
|
269
|
+
callback_function: (chunk: string) => {
|
|
270
|
+
partial += chunk
|
|
271
|
+
options.onPartial?.(cleanCaption(partial))
|
|
272
|
+
},
|
|
273
|
+
})
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const output = await model.generate({
|
|
277
|
+
...inputs,
|
|
278
|
+
max_new_tokens: maxNewTokens,
|
|
279
|
+
do_sample: false,
|
|
280
|
+
repetition_penalty: 1.05,
|
|
281
|
+
...(streamer ? { streamer } : {}),
|
|
282
|
+
})
|
|
283
|
+
|
|
284
|
+
const promptLength = (inputs.input_ids as Tensor).dims.at(-1) ?? 0
|
|
285
|
+
const decoded = processor.batch_decode(output.slice(null, [promptLength, null]), { skip_special_tokens: true })
|
|
286
|
+
const text = cleanCaption(decoded[0] ?? '')
|
|
287
|
+
|
|
288
|
+
this.#onProgress?.({ status: 'done' })
|
|
289
|
+
return { text }
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
dispose(): void {
|
|
293
|
+
const model = this.#model
|
|
294
|
+
this.#model = null
|
|
295
|
+
this.#processor = null
|
|
296
|
+
this.#resolvedDevice = null
|
|
297
|
+
void model?.dispose?.()
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
export async function caption(image: CaptionImageInput, options?: CaptionOptions): Promise<CaptionResult> {
|
|
302
|
+
const captioner = new ImageCaptioner(options)
|
|
303
|
+
try {
|
|
304
|
+
return await captioner.caption(image)
|
|
305
|
+
} finally {
|
|
306
|
+
captioner.dispose()
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
export type { ProgressInfo, Device }
|