runonweb 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -0
- package/package.json +126 -0
- package/src/caption/index.ts +310 -0
- package/src/clean/index.ts +263 -0
- package/src/core/cache.ts +248 -0
- package/src/core/device.ts +34 -0
- package/src/core/index.ts +18 -0
- package/src/core/pipeline.ts +108 -0
- package/src/core/progress.ts +20 -0
- package/src/depth/index.ts +127 -0
- package/src/detect/index.ts +274 -0
- package/src/embed/index.ts +172 -0
- package/src/emoji/index.ts +132 -0
- package/src/image/engine.d.ts +57 -0
- package/src/image/engine.js +14090 -0
- package/src/image/index.ts +230 -0
- package/src/image/sizes.ts +28 -0
- package/src/ocr/index.ts +416 -0
- package/src/ocr/sizes.ts +12 -0
- package/src/remove-bg/index.ts +170 -0
- package/src/stt/index.ts +320 -0
- package/src/translate/bergamot.ts +280 -0
- package/src/translate/index.ts +240 -0
- package/src/translate/registry.ts +133 -0
- package/src/translate/worker.ts +181 -0
- package/src/tts/index.ts +220 -0
- package/src/tts/kitten.ts +280 -0
- package/src/tts/kokoro.ts +140 -0
- package/src/tts/phonemes.ts +150 -0
- package/src/tts/sizes.ts +30 -0
- package/src/tts/split.ts +31 -0
- package/src/tts/supertonic.ts +126 -0
- package/src/tts/types.ts +11 -0
- package/src/tts/voices.ts +131 -0
- package/src/tts/wav.ts +52 -0
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
import type { ProgressCallback } from '../core/index.ts'
|
|
2
|
+
import { splitUtterances } from './split.ts'
|
|
3
|
+
import { toFloat32 } from './wav.ts'
|
|
4
|
+
import type { SpeakChunk } from './types.ts'
|
|
5
|
+
|
|
6
|
+
const HF = 'https://huggingface.co/KittenML/kitten-tts-nano-0.8-int8/resolve/main'
|
|
7
|
+
const CACHE = 'runonweb-kitten-tts'
|
|
8
|
+
const ORT_WASM = 'https://cdn.jsdelivr.net/npm/onnxruntime-web@1.30.0/dist/'
|
|
9
|
+
const SAMPLE_RATE = 24_000
|
|
10
|
+
const AUDIO_TRIM = 5_000
|
|
11
|
+
|
|
12
|
+
const VOICE_ALIASES: Record<string, string> = {
|
|
13
|
+
bella: 'expr-voice-2-f',
|
|
14
|
+
jasper: 'expr-voice-2-m',
|
|
15
|
+
luna: 'expr-voice-3-f',
|
|
16
|
+
bruno: 'expr-voice-3-m',
|
|
17
|
+
rosie: 'expr-voice-4-f',
|
|
18
|
+
hugo: 'expr-voice-4-m',
|
|
19
|
+
kiki: 'expr-voice-5-f',
|
|
20
|
+
leo: 'expr-voice-5-m',
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const SPEED_PRIORS: Record<string, number> = {
|
|
24
|
+
'expr-voice-2-f': 0.8,
|
|
25
|
+
'expr-voice-2-m': 0.8,
|
|
26
|
+
'expr-voice-3-m': 0.8,
|
|
27
|
+
'expr-voice-3-f': 0.8,
|
|
28
|
+
'expr-voice-4-m': 0.9,
|
|
29
|
+
'expr-voice-4-f': 0.8,
|
|
30
|
+
'expr-voice-5-m': 0.8,
|
|
31
|
+
'expr-voice-5-f': 0.8,
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const PAD = '$'
|
|
35
|
+
const PUNCTUATION = ';:,.!?¡¿—…"«»"" '
|
|
36
|
+
const LETTERS = 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz'
|
|
37
|
+
const LETTERS_IPA =
|
|
38
|
+
"ɑɐɒæɓʙβɔɕçɗɖðʤəɘɚɛɜɝɞɟʄɡɠɢʛɦɧħɥʜɨɪʝɭɬɫɮʟɱɯɰŋɳɲɴøɵɸθœɶʘɹɺɾɻʀʁɽʂʃʈʧʉʊʋⱱʌɣɤʍχʎʏʑʐʒʔʡʕʢǀǁǂǃˈˌːˑʼʴʰʱʲʷˠˤ˞↓↑→↗↘''ᵻ"
|
|
39
|
+
const SYMBOLS = [PAD, ...PUNCTUATION, ...LETTERS, ...LETTERS_IPA]
|
|
40
|
+
const SYMBOL_TO_ID = new Map(SYMBOLS.map((s, i) => [s, i]))
|
|
41
|
+
|
|
42
|
+
type OrtApi = {
|
|
43
|
+
env: { wasm: { wasmPaths: string; numThreads: number; simd?: boolean } }
|
|
44
|
+
Tensor: new (type: string, data: Float32Array | BigInt64Array, dims: number[]) => unknown
|
|
45
|
+
InferenceSession: {
|
|
46
|
+
create: (
|
|
47
|
+
model: ArrayBuffer,
|
|
48
|
+
options?: { executionProviders?: string[] }
|
|
49
|
+
) => Promise<{ run: (feeds: Record<string, unknown>) => Promise<Record<string, { data: Float32Array }>> }>
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
type VoiceEntry = { data: Float32Array; shape: number[] }
|
|
54
|
+
|
|
55
|
+
export type LoadedKitten = {
|
|
56
|
+
session: Awaited<ReturnType<OrtApi['InferenceSession']['create']>>
|
|
57
|
+
voices: Record<string, VoiceEntry>
|
|
58
|
+
ort: OrtApi
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export async function loadKitten(onProgress?: ProgressCallback): Promise<LoadedKitten> {
|
|
62
|
+
onProgress?.({ status: 'loading', progress: 0 })
|
|
63
|
+
const [model, voicesBuf] = await Promise.all([
|
|
64
|
+
fetchCached(`${HF}/kitten_tts_nano_v0_8.onnx`, 'kitten.onnx', onProgress),
|
|
65
|
+
fetchCached(`${HF}/voices.npz`, 'voices.npz', onProgress),
|
|
66
|
+
])
|
|
67
|
+
const [ort, voices] = await Promise.all([loadOrt(), loadNpz(voicesBuf)])
|
|
68
|
+
const session = await ort.InferenceSession.create(model, { executionProviders: ['wasm'] })
|
|
69
|
+
onProgress?.({ status: 'ready', progress: 100 })
|
|
70
|
+
return { session, voices, ort }
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export async function* streamKitten(
|
|
74
|
+
loaded: LoadedKitten,
|
|
75
|
+
text: string,
|
|
76
|
+
voice: string,
|
|
77
|
+
speed: number
|
|
78
|
+
): AsyncGenerator<SpeakChunk> {
|
|
79
|
+
for (const piece of splitUtterances(text)) {
|
|
80
|
+
const audio = await inferKitten(loaded, piece, voice, speed)
|
|
81
|
+
yield { audio, samplingRate: SAMPLE_RATE, text: piece }
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export function disposeKitten(loaded: LoadedKitten | null) {
|
|
86
|
+
const session = loaded?.session as { release?: () => Promise<void> } | undefined
|
|
87
|
+
void session?.release?.()
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
async function inferKitten(loaded: LoadedKitten, text: string, voiceId: string, speed: number): Promise<Float32Array> {
|
|
91
|
+
const key = VOICE_ALIASES[voiceId] ?? voiceId
|
|
92
|
+
const voice = loaded.voices[key]
|
|
93
|
+
if (!voice) throw new Error(`Unknown Kitten voice: ${voiceId}`)
|
|
94
|
+
|
|
95
|
+
const phonemes = await phonemizeEnglish(ensurePunctuation(text))
|
|
96
|
+
const tokenIds = cleanPhonemes(phonemes)
|
|
97
|
+
const [numStyles, styleDim] = voice.shape
|
|
98
|
+
const refId = Math.min(tokenIds.length, (numStyles ?? 1) - 1)
|
|
99
|
+
const dim = styleDim ?? 256
|
|
100
|
+
const style = voice.data.slice(refId * dim, (refId + 1) * dim)
|
|
101
|
+
const scaled = speed * (SPEED_PRIORS[key] ?? 1)
|
|
102
|
+
|
|
103
|
+
const feeds = {
|
|
104
|
+
input_ids: new loaded.ort.Tensor('int64', BigInt64Array.from(tokenIds.map(BigInt)), [1, tokenIds.length]),
|
|
105
|
+
style: new loaded.ort.Tensor('float32', new Float32Array(style), [1, dim]),
|
|
106
|
+
speed: new loaded.ort.Tensor('float32', new Float32Array([scaled]), [1]),
|
|
107
|
+
}
|
|
108
|
+
const results = await loaded.session.run(feeds)
|
|
109
|
+
const output = results[Object.keys(results)[0] ?? '']
|
|
110
|
+
const data = toFloat32(output?.data ?? [])
|
|
111
|
+
return data.slice(0, Math.max(0, data.length - AUDIO_TRIM))
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
function cleanPhonemes(phonemes: string): number[] {
|
|
115
|
+
const ids: number[] = []
|
|
116
|
+
for (const ch of phonemes) {
|
|
117
|
+
const id = SYMBOL_TO_ID.get(ch)
|
|
118
|
+
if (id !== undefined) ids.push(id)
|
|
119
|
+
}
|
|
120
|
+
return [0, ...ids, 10, 0]
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function ensurePunctuation(text: string): string {
|
|
124
|
+
const t = text.trim()
|
|
125
|
+
if (!t) return t
|
|
126
|
+
return /[.!?,;:]$/.test(t) ? t : `${t},`
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
async function phonemizeEnglish(text: string): Promise<string> {
|
|
130
|
+
const { phonemize } = await import('phonemizer')
|
|
131
|
+
const chunks = text.split(/([;:,.!?¡¿—…"«»"()\n]+)/)
|
|
132
|
+
let out = ''
|
|
133
|
+
for (const chunk of chunks) {
|
|
134
|
+
if (!chunk) continue
|
|
135
|
+
if (/^[;:,.!?¡¿—…"«»"()\s]+$/.test(chunk)) {
|
|
136
|
+
out += chunk
|
|
137
|
+
continue
|
|
138
|
+
}
|
|
139
|
+
const ipa = await phonemize(chunk, 'en-us')
|
|
140
|
+
out += ipa.join(' ').replace(/_/g, '')
|
|
141
|
+
}
|
|
142
|
+
return out.trim()
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
async function loadOrt(): Promise<OrtApi> {
|
|
146
|
+
const ort = (await import('onnxruntime-web')) as unknown as OrtApi
|
|
147
|
+
ort.env.wasm.wasmPaths = ORT_WASM
|
|
148
|
+
ort.env.wasm.numThreads = 1
|
|
149
|
+
return ort
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
async function fetchCached(url: string, file: string, onProgress?: ProgressCallback): Promise<ArrayBuffer> {
|
|
153
|
+
const cache = await caches.open(CACHE).catch(() => null)
|
|
154
|
+
const hit = await cache?.match(url)
|
|
155
|
+
if (hit) {
|
|
156
|
+
onProgress?.({ status: 'progress', progress: 100, file })
|
|
157
|
+
return hit.arrayBuffer()
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
const res = await fetch(url)
|
|
161
|
+
if (!res.ok) throw new Error(`Could not load ${file} (${res.status})`)
|
|
162
|
+
|
|
163
|
+
const total = Number(res.headers.get('content-length') ?? 0)
|
|
164
|
+
if (!res.body || !total) {
|
|
165
|
+
const buf = await res.arrayBuffer()
|
|
166
|
+
await cache?.put(url, new Response(buf.slice(0)))
|
|
167
|
+
return buf
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
const reader = res.body.getReader()
|
|
171
|
+
const chunks: Uint8Array[] = []
|
|
172
|
+
let received = 0
|
|
173
|
+
for (;;) {
|
|
174
|
+
const { done, value } = await reader.read()
|
|
175
|
+
if (done) break
|
|
176
|
+
chunks.push(value)
|
|
177
|
+
received += value.byteLength
|
|
178
|
+
onProgress?.({ status: 'progress', progress: (received / total) * 100, file })
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const out = new Uint8Array(received)
|
|
182
|
+
let offset = 0
|
|
183
|
+
for (const chunk of chunks) {
|
|
184
|
+
out.set(chunk, offset)
|
|
185
|
+
offset += chunk.byteLength
|
|
186
|
+
}
|
|
187
|
+
await cache?.put(url, new Response(out.buffer.slice(0)))
|
|
188
|
+
return out.buffer
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
async function loadNpz(buffer: ArrayBuffer): Promise<Record<string, VoiceEntry>> {
|
|
192
|
+
const view = new DataView(buffer)
|
|
193
|
+
const bytes = new Uint8Array(buffer)
|
|
194
|
+
const voices: Record<string, VoiceEntry> = {}
|
|
195
|
+
|
|
196
|
+
const eocd = findEocd(view, bytes.length)
|
|
197
|
+
const count = view.getUint16(eocd + 10, true)
|
|
198
|
+
let central = view.getUint32(eocd + 16, true)
|
|
199
|
+
|
|
200
|
+
for (let i = 0; i < count; i++) {
|
|
201
|
+
if (view.getUint32(central, true) !== 0x02014b50) break
|
|
202
|
+
const method = view.getUint16(central + 10, true)
|
|
203
|
+
const compSize = view.getUint32(central + 20, true)
|
|
204
|
+
const rawSize = view.getUint32(central + 24, true)
|
|
205
|
+
const nameLen = view.getUint16(central + 28, true)
|
|
206
|
+
const extraLen = view.getUint16(central + 30, true)
|
|
207
|
+
const commentLen = view.getUint16(central + 32, true)
|
|
208
|
+
const localOff = view.getUint32(central + 42, true)
|
|
209
|
+
const name = new TextDecoder().decode(bytes.subarray(central + 46, central + 46 + nameLen))
|
|
210
|
+
central += 46 + nameLen + extraLen + commentLen
|
|
211
|
+
if (!name.endsWith('.npy')) continue
|
|
212
|
+
|
|
213
|
+
const localNameLen = view.getUint16(localOff + 26, true)
|
|
214
|
+
const localExtraLen = view.getUint16(localOff + 28, true)
|
|
215
|
+
const dataStart = localOff + 30 + localNameLen + localExtraLen
|
|
216
|
+
const compressed = bytes.subarray(dataStart, dataStart + compSize)
|
|
217
|
+
const raw =
|
|
218
|
+
method === 0 ? copyBytes(compressed) : method === 8 ? await inflateRaw(compressed, rawSize) : null
|
|
219
|
+
if (!raw) throw new Error(`Unsupported zip method ${method} in voices.npz`)
|
|
220
|
+
const key = name.replace(/\.npy$/, '').split('/').pop() ?? name
|
|
221
|
+
voices[key] = parseNpy(copyBytes(raw).buffer)
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
if (!Object.keys(voices).length) throw new Error('voices.npz contained no arrays')
|
|
225
|
+
return voices
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
function findEocd(view: DataView, length: number): number {
|
|
229
|
+
for (let i = length - 22; i >= Math.max(0, length - 22 - 0xffff); i--) {
|
|
230
|
+
if (view.getUint32(i, true) === 0x06054b50) return i
|
|
231
|
+
}
|
|
232
|
+
throw new Error('voices.npz is not a valid zip')
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
function copyBytes(src: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
236
|
+
const out = new Uint8Array(new ArrayBuffer(src.byteLength))
|
|
237
|
+
out.set(src)
|
|
238
|
+
return out
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
async function inflateRaw(data: Uint8Array, outLen: number): Promise<Uint8Array> {
|
|
242
|
+
const stream = new DecompressionStream('deflate-raw')
|
|
243
|
+
const writer = stream.writable.getWriter()
|
|
244
|
+
await writer.write(Uint8Array.from(data))
|
|
245
|
+
await writer.close()
|
|
246
|
+
const reader = stream.readable.getReader()
|
|
247
|
+
const chunks: Uint8Array[] = []
|
|
248
|
+
let received = 0
|
|
249
|
+
for (;;) {
|
|
250
|
+
const { done, value } = await reader.read()
|
|
251
|
+
if (done) break
|
|
252
|
+
chunks.push(value)
|
|
253
|
+
received += value.byteLength
|
|
254
|
+
}
|
|
255
|
+
const out = new Uint8Array(outLen || received)
|
|
256
|
+
let offset = 0
|
|
257
|
+
for (const chunk of chunks) {
|
|
258
|
+
out.set(chunk, offset)
|
|
259
|
+
offset += chunk.byteLength
|
|
260
|
+
}
|
|
261
|
+
return out
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
function parseNpy(buf: ArrayBuffer): VoiceEntry {
|
|
265
|
+
const bytes = new Uint8Array(buf)
|
|
266
|
+
const major = bytes[6] ?? 1
|
|
267
|
+
const headerLen =
|
|
268
|
+
major >= 2 ? new DataView(buf, 8, 4).getUint32(0, true) : new DataView(buf, 8, 2).getUint16(0, true)
|
|
269
|
+
const headerOffset = major >= 2 ? 12 : 10
|
|
270
|
+
const header = new TextDecoder().decode(bytes.subarray(headerOffset, headerOffset + headerLen))
|
|
271
|
+
const shapeStr = header.match(/'shape'\s*:\s*\(([^)]*)\)/)?.[1]?.trim() ?? ''
|
|
272
|
+
const shape = shapeStr === '' ? [1] : shapeStr.split(',').map((s) => Number(s.trim())).filter((n) => !Number.isNaN(n))
|
|
273
|
+
const start = headerOffset + headerLen
|
|
274
|
+
const avail = buf.byteLength - start
|
|
275
|
+
const usable = avail - (avail % 4)
|
|
276
|
+
const data = new Float32Array(buf.slice(start, start + usable))
|
|
277
|
+
return { data, shape }
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
export { SAMPLE_RATE as KITTEN_SAMPLE_RATE }
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import type { Device, ProgressCallback, ResolvedDevice } from '../core/index.ts'
|
|
2
|
+
import { resolveDevice, toProgressInfo } from '../core/index.ts'
|
|
3
|
+
import { phonemizeLang } from './phonemes.ts'
|
|
4
|
+
import { splitUtterances } from './split.ts'
|
|
5
|
+
import { toFloat32 } from './wav.ts'
|
|
6
|
+
import { kokoroPhonemeLang } from './voices.ts'
|
|
7
|
+
import type { SpeakChunk } from './types.ts'
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Kokoro 82M (StyleTTS 2) on Transformers.js directly. No `kokoro-js`, so the app ships one
|
|
11
|
+
* copy of Transformers.js. Voice style vectors come from the same Hugging Face repo.
|
|
12
|
+
*/
|
|
13
|
+
const DEFAULT_MODEL = 'onnx-community/Kokoro-82M-v1.0-ONNX'
|
|
14
|
+
const SAMPLE_RATE = 24_000
|
|
15
|
+
const STYLE_DIM = 256
|
|
16
|
+
const MAX_STYLE_INDEX = 509
|
|
17
|
+
const VOICE_CACHE = 'runonweb-kokoro-voices'
|
|
18
|
+
|
|
19
|
+
type Tensorish = { data: Float32Array | number[]; dims: number[] }
|
|
20
|
+
|
|
21
|
+
type KokoroModel = {
|
|
22
|
+
(inputs: Record<string, unknown>): Promise<{ waveform: Tensorish }>
|
|
23
|
+
dispose?: () => Promise<void>
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
type KokoroTokenizer = (text: string, options?: { truncation?: boolean }) => { input_ids: { dims: number[] } }
|
|
27
|
+
|
|
28
|
+
export type LoadedKokoro = {
|
|
29
|
+
model: KokoroModel
|
|
30
|
+
tokenizer: KokoroTokenizer
|
|
31
|
+
modelId: string
|
|
32
|
+
device: ResolvedDevice
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export async function loadKokoro(options: {
|
|
36
|
+
model?: string
|
|
37
|
+
device?: Device
|
|
38
|
+
onProgress?: ProgressCallback
|
|
39
|
+
}): Promise<LoadedKokoro> {
|
|
40
|
+
const device = await resolveDevice(options.device ?? 'auto')
|
|
41
|
+
options.onProgress?.({ status: 'loading', progress: 0 })
|
|
42
|
+
|
|
43
|
+
const { StyleTextToSpeech2Model, AutoTokenizer, env } = await import('@huggingface/transformers')
|
|
44
|
+
env.allowLocalModels = false
|
|
45
|
+
const modelId = options.model ?? DEFAULT_MODEL
|
|
46
|
+
const progress_callback = (data: Record<string, unknown>) => {
|
|
47
|
+
options.onProgress?.(toProgressInfo(data))
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const [model, tokenizer] = await Promise.all([
|
|
51
|
+
StyleTextToSpeech2Model.from_pretrained(modelId, {
|
|
52
|
+
device,
|
|
53
|
+
// fp16 / q4f16 produce NaN on the Transformers.js 4 WebGPU runtime; fp32 is clean and fast.
|
|
54
|
+
dtype: device === 'webgpu' ? 'fp32' : 'q8',
|
|
55
|
+
progress_callback,
|
|
56
|
+
}),
|
|
57
|
+
AutoTokenizer.from_pretrained(modelId, { progress_callback }),
|
|
58
|
+
])
|
|
59
|
+
|
|
60
|
+
options.onProgress?.({ status: 'ready', progress: 100 })
|
|
61
|
+
return {
|
|
62
|
+
model: model as unknown as KokoroModel,
|
|
63
|
+
tokenizer: tokenizer as unknown as KokoroTokenizer,
|
|
64
|
+
modelId,
|
|
65
|
+
device,
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export async function* streamKokoro(
|
|
70
|
+
kokoro: LoadedKokoro,
|
|
71
|
+
text: string,
|
|
72
|
+
voice: string,
|
|
73
|
+
speed: number
|
|
74
|
+
): AsyncGenerator<SpeakChunk> {
|
|
75
|
+
const lang = kokoroPhonemeLang(voice) ?? (voice.startsWith('b') ? 'en-gb' : 'en-us')
|
|
76
|
+
for (const piece of splitUtterances(text)) {
|
|
77
|
+
const phonemes = await phonemizeLang(piece, lang)
|
|
78
|
+
if (!phonemes) throw new Error(`Could not phonemize text for ${lang}`)
|
|
79
|
+
const audio = await synthesize(kokoro, phonemes, voice, speed)
|
|
80
|
+
yield { audio, samplingRate: SAMPLE_RATE, text: piece }
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
async function synthesize(kokoro: LoadedKokoro, phonemes: string, voice: string, speed: number): Promise<Float32Array> {
|
|
85
|
+
const { Tensor } = await import('@huggingface/transformers')
|
|
86
|
+
const { input_ids } = kokoro.tokenizer(phonemes, { truncation: true })
|
|
87
|
+
const numTokens = Math.min(Math.max((input_ids.dims.at(-1) ?? 2) - 2, 0), MAX_STYLE_INDEX)
|
|
88
|
+
const styles = await loadVoice(kokoro.modelId, voice)
|
|
89
|
+
const style = styles.slice(numTokens * STYLE_DIM, (numTokens + 1) * STYLE_DIM)
|
|
90
|
+
|
|
91
|
+
const { waveform } = await kokoro.model({
|
|
92
|
+
input_ids,
|
|
93
|
+
style: new Tensor('float32', style, [1, STYLE_DIM]),
|
|
94
|
+
speed: new Tensor('float32', [speed], [1]),
|
|
95
|
+
})
|
|
96
|
+
return toFloat32(waveform.data)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
const voiceCache = new Map<string, Promise<Float32Array>>()
|
|
100
|
+
|
|
101
|
+
/** Fetch `voices/<id>.bin` (510×256 float32 styles indexed by token count) and keep it in the Cache API. */
|
|
102
|
+
function loadVoice(modelId: string, voice: string): Promise<Float32Array> {
|
|
103
|
+
const key = `${modelId}/${voice}`
|
|
104
|
+
let pending = voiceCache.get(key)
|
|
105
|
+
if (!pending) {
|
|
106
|
+
pending = fetchVoice(modelId, voice).catch((err) => {
|
|
107
|
+
voiceCache.delete(key)
|
|
108
|
+
throw err
|
|
109
|
+
})
|
|
110
|
+
voiceCache.set(key, pending)
|
|
111
|
+
}
|
|
112
|
+
return pending
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
async function fetchVoice(modelId: string, voice: string): Promise<Float32Array> {
|
|
116
|
+
const url = `https://huggingface.co/${modelId}/resolve/main/voices/${voice}.bin`
|
|
117
|
+
let cache: Cache | null = null
|
|
118
|
+
try {
|
|
119
|
+
cache = await caches.open(VOICE_CACHE)
|
|
120
|
+
const hit = await cache.match(url)
|
|
121
|
+
if (hit) return new Float32Array(await hit.arrayBuffer())
|
|
122
|
+
} catch {
|
|
123
|
+
cache = null
|
|
124
|
+
}
|
|
125
|
+
const res = await fetch(url)
|
|
126
|
+
if (!res.ok) throw new Error(`Voice "${voice}" not found (${res.status})`)
|
|
127
|
+
const buffer = await res.arrayBuffer()
|
|
128
|
+
try {
|
|
129
|
+
await cache?.put(url, new Response(buffer.slice(0), { headers: { 'content-type': 'application/octet-stream' } }))
|
|
130
|
+
} catch {
|
|
131
|
+
// Cache is best-effort.
|
|
132
|
+
}
|
|
133
|
+
return new Float32Array(buffer)
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
export function disposeKokoro(kokoro: LoadedKokoro | null) {
|
|
137
|
+
void kokoro?.model.dispose?.()
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export { DEFAULT_MODEL as KOKORO_MODEL, SAMPLE_RATE as KOKORO_SAMPLE_RATE }
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
import { kokoroPhonemeLang } from './voices.ts'
|
|
2
|
+
|
|
3
|
+
const ESPEAK_JS = 'https://cdn.jsdelivr.net/npm/espeak-ng@1.0.2/dist/espeak-ng.js'
|
|
4
|
+
const ESPEAK_WASM = 'https://cdn.jsdelivr.net/npm/espeak-ng@1.0.2/dist/espeak-ng.wasm'
|
|
5
|
+
|
|
6
|
+
type ESpeakInstance = {
|
|
7
|
+
FS: { readFile: (name: string, opts: { encoding: string }) => string }
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
type ESpeakFactory = (opts: {
|
|
11
|
+
locateFile?: (file: string) => string
|
|
12
|
+
arguments?: string[]
|
|
13
|
+
}) => Promise<ESpeakInstance>
|
|
14
|
+
|
|
15
|
+
let factory: ESpeakFactory | null = null
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Convert text to IPA for Kokoro. English uses `phonemizer` with Kokoro's text normalization
|
|
19
|
+
* and phoneme fixes (ported from kokoro-js); ES/FR load eSpeak-NG WASM on demand.
|
|
20
|
+
*/
|
|
21
|
+
export async function phonemizeLang(text: string, lang: string): Promise<string> {
|
|
22
|
+
if (lang === 'en-us' || lang === 'en-gb' || lang === 'en') {
|
|
23
|
+
return phonemizeEnglish(text, lang === 'en-us' ? 'a' : 'b')
|
|
24
|
+
}
|
|
25
|
+
return phonemizeEspeak(text, lang)
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
const PUNCT = ';:,.!?¡¿—…"«»“”(){}[]'
|
|
29
|
+
const PUNCT_RE = new RegExp(`(\\s*[${PUNCT.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}]+\\s*)+`, 'g')
|
|
30
|
+
|
|
31
|
+
async function phonemizeEnglish(text: string, variant: 'a' | 'b'): Promise<string> {
|
|
32
|
+
const { phonemize } = await import('phonemizer')
|
|
33
|
+
const normalized = normalizeEnglish(text)
|
|
34
|
+
const lang = variant === 'a' ? 'en-us' : 'en'
|
|
35
|
+
|
|
36
|
+
// Phonemize the text between punctuation runs; keep the punctuation verbatim.
|
|
37
|
+
const parts: string[] = []
|
|
38
|
+
let last = 0
|
|
39
|
+
for (const m of normalized.matchAll(PUNCT_RE)) {
|
|
40
|
+
const idx = m.index ?? 0
|
|
41
|
+
if (last < idx) parts.push((await phonemize(normalized.slice(last, idx), lang)).join(' '))
|
|
42
|
+
if (m[0].length > 0) parts.push(m[0])
|
|
43
|
+
last = idx + m[0].length
|
|
44
|
+
}
|
|
45
|
+
if (last < normalized.length) parts.push((await phonemize(normalized.slice(last), lang)).join(' '))
|
|
46
|
+
|
|
47
|
+
let ps = parts
|
|
48
|
+
.join('')
|
|
49
|
+
.replace(/kəkˈoːɹoʊ/g, 'kˈoʊkəɹoʊ')
|
|
50
|
+
.replace(/kəkˈɔːɹəʊ/g, 'kˈəʊkəɹəʊ')
|
|
51
|
+
.replace(/ʲ/g, 'j')
|
|
52
|
+
.replace(/r/g, 'ɹ')
|
|
53
|
+
.replace(/x/g, 'k')
|
|
54
|
+
.replace(/ɬ/g, 'l')
|
|
55
|
+
.replace(/(?<=[a-zɹː])(?=hˈʌndɹɪd)/g, ' ')
|
|
56
|
+
.replace(/ z(?=[;:,.!?¡¿—…"«»“” ]|$)/g, 'z')
|
|
57
|
+
if (variant === 'a') ps = ps.replace(/(?<=nˈaɪn)ti(?!ː)/g, 'di')
|
|
58
|
+
return ps.replace(/\s+/g, ' ').trim()
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function normalizeEnglish(text: string): string {
|
|
62
|
+
return text
|
|
63
|
+
.replace(/[‘’]/g, "'")
|
|
64
|
+
.replace(/«/g, '“')
|
|
65
|
+
.replace(/»/g, '”')
|
|
66
|
+
.replace(/[“”]/g, '"')
|
|
67
|
+
.replace(/\(/g, '«')
|
|
68
|
+
.replace(/\)/g, '»')
|
|
69
|
+
.replace(/、/g, ', ')
|
|
70
|
+
.replace(/。/g, '. ')
|
|
71
|
+
.replace(/!/g, '! ')
|
|
72
|
+
.replace(/,/g, ', ')
|
|
73
|
+
.replace(/:/g, ': ')
|
|
74
|
+
.replace(/;/g, '; ')
|
|
75
|
+
.replace(/?/g, '? ')
|
|
76
|
+
.replace(/[^\S \n]/g, ' ')
|
|
77
|
+
.replace(/ +/, ' ')
|
|
78
|
+
.replace(/(?<=\n) +(?=\n)/g, '')
|
|
79
|
+
.replace(/\bD[Rr]\.(?= [A-Z])/g, 'Doctor')
|
|
80
|
+
.replace(/\b(?:Mr\.|MR\.(?= [A-Z]))/g, 'Mister')
|
|
81
|
+
.replace(/\b(?:Ms\.|MS\.(?= [A-Z]))/g, 'Miss')
|
|
82
|
+
.replace(/\b(?:Mrs\.|MRS\.(?= [A-Z]))/g, 'Mrs')
|
|
83
|
+
.replace(/\betc\.(?! [A-Z])/gi, 'etc')
|
|
84
|
+
.replace(/\b(y)eah?\b/gi, "$1e'a")
|
|
85
|
+
.replace(/\d*\.\d+|\b\d{4}s?\b|(?<!:)\b(?:[1-9]|1[0-2]):[0-5]\d\b(?!:)/g, splitNum)
|
|
86
|
+
.replace(/(?<=\d),(?=\d)/g, '')
|
|
87
|
+
.replace(/[$£]\d+(?:\.\d+)?(?: hundred| thousand| (?:[bm]|tr)illion)*\b|[$£]\d+\.\d\d?\b/gi, flipMoney)
|
|
88
|
+
.replace(/\d*\.\d+/g, pointNum)
|
|
89
|
+
.replace(/(?<=\d)-(?=\d)/g, ' to ')
|
|
90
|
+
.replace(/(?<=\d)S/g, ' S')
|
|
91
|
+
.replace(/(?<=[BCDFGHJ-NP-TV-Z])'?s\b/g, "'S")
|
|
92
|
+
.replace(/(?<=X')S\b/g, 's')
|
|
93
|
+
.replace(/(?:[A-Za-z]\.){2,} [a-z]/g, (m) => m.replace(/\./g, '-'))
|
|
94
|
+
.replace(/(?<=[A-Z])\.(?=[A-Z])/gi, '-')
|
|
95
|
+
.trim()
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function splitNum(match: string): string {
|
|
99
|
+
if (match.includes('.')) return match
|
|
100
|
+
if (match.includes(':')) {
|
|
101
|
+
const [h, m] = match.split(':').map(Number)
|
|
102
|
+
if (m === 0) return `${h} o'clock`
|
|
103
|
+
if (m! < 10) return `${h} oh ${m}`
|
|
104
|
+
return `${h} ${m}`
|
|
105
|
+
}
|
|
106
|
+
const year = parseInt(match.slice(0, 4), 10)
|
|
107
|
+
if (year < 1100 || year % 1000 < 10) return match
|
|
108
|
+
const left = match.slice(0, 2)
|
|
109
|
+
const right = parseInt(match.slice(2, 4), 10)
|
|
110
|
+
const suffix = match.endsWith('s') ? 's' : ''
|
|
111
|
+
if (year % 1000 >= 100 && year % 1000 <= 999) {
|
|
112
|
+
if (right === 0) return `${left} hundred${suffix}`
|
|
113
|
+
if (right < 10) return `${left} oh ${right}${suffix}`
|
|
114
|
+
}
|
|
115
|
+
return `${left} ${right}${suffix}`
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function flipMoney(match: string): string {
|
|
119
|
+
const unit = match[0] === '$' ? 'dollar' : 'pound'
|
|
120
|
+
if (isNaN(Number(match.slice(1)))) return `${match.slice(1)} ${unit}s`
|
|
121
|
+
if (!match.includes('.')) {
|
|
122
|
+
const s = match.slice(1) === '1' ? '' : 's'
|
|
123
|
+
return `${match.slice(1)} ${unit}${s}`
|
|
124
|
+
}
|
|
125
|
+
const [whole, frac] = match.slice(1).split('.')
|
|
126
|
+
const cents = parseInt((frac ?? '').padEnd(2, '0'), 10)
|
|
127
|
+
const centUnit = match[0] === '$' ? (cents === 1 ? 'cent' : 'cents') : cents === 1 ? 'penny' : 'pence'
|
|
128
|
+
return `${whole} ${unit}${whole === '1' ? '' : 's'} and ${cents} ${centUnit}`
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function pointNum(match: string): string {
|
|
132
|
+
const [whole, frac] = match.split('.')
|
|
133
|
+
return `${whole} point ${(frac ?? '').split('').join(' ')}`
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
async function phonemizeEspeak(text: string, lang: string): Promise<string> {
|
|
137
|
+
if (!factory) {
|
|
138
|
+
const mod = (await import(/* @vite-ignore */ ESPEAK_JS)) as { default: ESpeakFactory }
|
|
139
|
+
factory = mod.default
|
|
140
|
+
}
|
|
141
|
+
const instance = await factory({
|
|
142
|
+
locateFile: (file) => (file.endsWith('.wasm') ? ESPEAK_WASM : file),
|
|
143
|
+
arguments: ['--phonout', 'generated', '-q', '-b=1', '--ipa=3', '-v', lang, text],
|
|
144
|
+
})
|
|
145
|
+
return instance.FS.readFile('generated', { encoding: 'utf8' }).replace(/\s+/g, ' ').trim()
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function needsExternalPhonemes(voiceId: string): boolean {
|
|
149
|
+
return kokoroPhonemeLang(voiceId) != null
|
|
150
|
+
}
|
package/src/tts/sizes.ts
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
export type TTSSize = 'tiny' | 'small' | 'multi'
|
|
2
|
+
|
|
3
|
+
export const TTS_SIZES: Record<
|
|
4
|
+
TTSSize,
|
|
5
|
+
{ label: string; engine: string; params: string; downloadMB: string; quality: string }
|
|
6
|
+
> = {
|
|
7
|
+
tiny: {
|
|
8
|
+
label: 'Tiny',
|
|
9
|
+
engine: 'KittenTTS',
|
|
10
|
+
params: '15M',
|
|
11
|
+
downloadMB: '~28 MB',
|
|
12
|
+
quality: '8 English voices',
|
|
13
|
+
},
|
|
14
|
+
small: {
|
|
15
|
+
label: 'Small',
|
|
16
|
+
engine: 'Kokoro 82M',
|
|
17
|
+
params: '82M',
|
|
18
|
+
downloadMB: '~326 MB · WASM ~92 MB',
|
|
19
|
+
quality: 'English · Spanish · French',
|
|
20
|
+
},
|
|
21
|
+
multi: {
|
|
22
|
+
label: 'Multi',
|
|
23
|
+
engine: 'Supertonic 2',
|
|
24
|
+
params: '66M',
|
|
25
|
+
downloadMB: '~262 MB',
|
|
26
|
+
quality: 'en · ko · es · pt · fr · 44.1 kHz',
|
|
27
|
+
},
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export const DEFAULT_TTS_SIZE: TTSSize = 'small'
|
package/src/tts/split.ts
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
const MAX_CHARS = 240
|
|
2
|
+
|
|
3
|
+
/** Split text into sentence-sized chunks for streaming synthesis. */
|
|
4
|
+
export function splitUtterances(text: string): string[] {
|
|
5
|
+
const trimmed = text.trim()
|
|
6
|
+
if (!trimmed) return []
|
|
7
|
+
|
|
8
|
+
const sentences = trimmed.split(/(?<=[.!?…])\s+|\n+/).map((s) => s.trim()).filter(Boolean)
|
|
9
|
+
const out: string[] = []
|
|
10
|
+
|
|
11
|
+
for (const sentence of sentences) {
|
|
12
|
+
if (sentence.length <= MAX_CHARS) {
|
|
13
|
+
out.push(sentence)
|
|
14
|
+
continue
|
|
15
|
+
}
|
|
16
|
+
const words = sentence.split(/\s+/)
|
|
17
|
+
let buf = ''
|
|
18
|
+
for (const word of words) {
|
|
19
|
+
const next = buf ? `${buf} ${word}` : word
|
|
20
|
+
if (next.length > MAX_CHARS && buf) {
|
|
21
|
+
out.push(buf)
|
|
22
|
+
buf = word
|
|
23
|
+
} else {
|
|
24
|
+
buf = next
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
if (buf) out.push(buf)
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
return out
|
|
31
|
+
}
|