runonweb 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,280 @@
1
+ import type { ProgressCallback } from '../core/index.ts'
2
+ import { splitUtterances } from './split.ts'
3
+ import { toFloat32 } from './wav.ts'
4
+ import type { SpeakChunk } from './types.ts'
5
+
6
+ const HF = 'https://huggingface.co/KittenML/kitten-tts-nano-0.8-int8/resolve/main'
7
+ const CACHE = 'runonweb-kitten-tts'
8
+ const ORT_WASM = 'https://cdn.jsdelivr.net/npm/onnxruntime-web@1.30.0/dist/'
9
+ const SAMPLE_RATE = 24_000
10
+ const AUDIO_TRIM = 5_000
11
+
12
+ const VOICE_ALIASES: Record<string, string> = {
13
+ bella: 'expr-voice-2-f',
14
+ jasper: 'expr-voice-2-m',
15
+ luna: 'expr-voice-3-f',
16
+ bruno: 'expr-voice-3-m',
17
+ rosie: 'expr-voice-4-f',
18
+ hugo: 'expr-voice-4-m',
19
+ kiki: 'expr-voice-5-f',
20
+ leo: 'expr-voice-5-m',
21
+ }
22
+
23
+ const SPEED_PRIORS: Record<string, number> = {
24
+ 'expr-voice-2-f': 0.8,
25
+ 'expr-voice-2-m': 0.8,
26
+ 'expr-voice-3-m': 0.8,
27
+ 'expr-voice-3-f': 0.8,
28
+ 'expr-voice-4-m': 0.9,
29
+ 'expr-voice-4-f': 0.8,
30
+ 'expr-voice-5-m': 0.8,
31
+ 'expr-voice-5-f': 0.8,
32
+ }
33
+
34
+ const PAD = '$'
35
+ const PUNCTUATION = ';:,.!?¡¿—…"«»"" '
36
+ const LETTERS = 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz'
37
+ const LETTERS_IPA =
38
+ "ɑɐɒæɓʙβɔɕçɗɖðʤəɘɚɛɜɝɞɟʄɡɠɢʛɦɧħɥʜɨɪʝɭɬɫɮʟɱɯɰŋɳɲɴøɵɸθœɶʘɹɺɾɻʀʁɽʂʃʈʧʉʊʋⱱʌɣɤʍχʎʏʑʐʒʔʡʕʢǀǁǂǃˈˌːˑʼʴʰʱʲʷˠˤ˞↓↑→↗↘''ᵻ"
39
+ const SYMBOLS = [PAD, ...PUNCTUATION, ...LETTERS, ...LETTERS_IPA]
40
+ const SYMBOL_TO_ID = new Map(SYMBOLS.map((s, i) => [s, i]))
41
+
42
+ type OrtApi = {
43
+ env: { wasm: { wasmPaths: string; numThreads: number; simd?: boolean } }
44
+ Tensor: new (type: string, data: Float32Array | BigInt64Array, dims: number[]) => unknown
45
+ InferenceSession: {
46
+ create: (
47
+ model: ArrayBuffer,
48
+ options?: { executionProviders?: string[] }
49
+ ) => Promise<{ run: (feeds: Record<string, unknown>) => Promise<Record<string, { data: Float32Array }>> }>
50
+ }
51
+ }
52
+
53
+ type VoiceEntry = { data: Float32Array; shape: number[] }
54
+
55
+ export type LoadedKitten = {
56
+ session: Awaited<ReturnType<OrtApi['InferenceSession']['create']>>
57
+ voices: Record<string, VoiceEntry>
58
+ ort: OrtApi
59
+ }
60
+
61
+ export async function loadKitten(onProgress?: ProgressCallback): Promise<LoadedKitten> {
62
+ onProgress?.({ status: 'loading', progress: 0 })
63
+ const [model, voicesBuf] = await Promise.all([
64
+ fetchCached(`${HF}/kitten_tts_nano_v0_8.onnx`, 'kitten.onnx', onProgress),
65
+ fetchCached(`${HF}/voices.npz`, 'voices.npz', onProgress),
66
+ ])
67
+ const [ort, voices] = await Promise.all([loadOrt(), loadNpz(voicesBuf)])
68
+ const session = await ort.InferenceSession.create(model, { executionProviders: ['wasm'] })
69
+ onProgress?.({ status: 'ready', progress: 100 })
70
+ return { session, voices, ort }
71
+ }
72
+
73
+ export async function* streamKitten(
74
+ loaded: LoadedKitten,
75
+ text: string,
76
+ voice: string,
77
+ speed: number
78
+ ): AsyncGenerator<SpeakChunk> {
79
+ for (const piece of splitUtterances(text)) {
80
+ const audio = await inferKitten(loaded, piece, voice, speed)
81
+ yield { audio, samplingRate: SAMPLE_RATE, text: piece }
82
+ }
83
+ }
84
+
85
+ export function disposeKitten(loaded: LoadedKitten | null) {
86
+ const session = loaded?.session as { release?: () => Promise<void> } | undefined
87
+ void session?.release?.()
88
+ }
89
+
90
+ async function inferKitten(loaded: LoadedKitten, text: string, voiceId: string, speed: number): Promise<Float32Array> {
91
+ const key = VOICE_ALIASES[voiceId] ?? voiceId
92
+ const voice = loaded.voices[key]
93
+ if (!voice) throw new Error(`Unknown Kitten voice: ${voiceId}`)
94
+
95
+ const phonemes = await phonemizeEnglish(ensurePunctuation(text))
96
+ const tokenIds = cleanPhonemes(phonemes)
97
+ const [numStyles, styleDim] = voice.shape
98
+ const refId = Math.min(tokenIds.length, (numStyles ?? 1) - 1)
99
+ const dim = styleDim ?? 256
100
+ const style = voice.data.slice(refId * dim, (refId + 1) * dim)
101
+ const scaled = speed * (SPEED_PRIORS[key] ?? 1)
102
+
103
+ const feeds = {
104
+ input_ids: new loaded.ort.Tensor('int64', BigInt64Array.from(tokenIds.map(BigInt)), [1, tokenIds.length]),
105
+ style: new loaded.ort.Tensor('float32', new Float32Array(style), [1, dim]),
106
+ speed: new loaded.ort.Tensor('float32', new Float32Array([scaled]), [1]),
107
+ }
108
+ const results = await loaded.session.run(feeds)
109
+ const output = results[Object.keys(results)[0] ?? '']
110
+ const data = toFloat32(output?.data ?? [])
111
+ return data.slice(0, Math.max(0, data.length - AUDIO_TRIM))
112
+ }
113
+
114
+ function cleanPhonemes(phonemes: string): number[] {
115
+ const ids: number[] = []
116
+ for (const ch of phonemes) {
117
+ const id = SYMBOL_TO_ID.get(ch)
118
+ if (id !== undefined) ids.push(id)
119
+ }
120
+ return [0, ...ids, 10, 0]
121
+ }
122
+
123
+ function ensurePunctuation(text: string): string {
124
+ const t = text.trim()
125
+ if (!t) return t
126
+ return /[.!?,;:]$/.test(t) ? t : `${t},`
127
+ }
128
+
129
+ async function phonemizeEnglish(text: string): Promise<string> {
130
+ const { phonemize } = await import('phonemizer')
131
+ const chunks = text.split(/([;:,.!?¡¿—…"«»"()\n]+)/)
132
+ let out = ''
133
+ for (const chunk of chunks) {
134
+ if (!chunk) continue
135
+ if (/^[;:,.!?¡¿—…"«»"()\s]+$/.test(chunk)) {
136
+ out += chunk
137
+ continue
138
+ }
139
+ const ipa = await phonemize(chunk, 'en-us')
140
+ out += ipa.join(' ').replace(/_/g, '')
141
+ }
142
+ return out.trim()
143
+ }
144
+
145
+ async function loadOrt(): Promise<OrtApi> {
146
+ const ort = (await import('onnxruntime-web')) as unknown as OrtApi
147
+ ort.env.wasm.wasmPaths = ORT_WASM
148
+ ort.env.wasm.numThreads = 1
149
+ return ort
150
+ }
151
+
152
+ async function fetchCached(url: string, file: string, onProgress?: ProgressCallback): Promise<ArrayBuffer> {
153
+ const cache = await caches.open(CACHE).catch(() => null)
154
+ const hit = await cache?.match(url)
155
+ if (hit) {
156
+ onProgress?.({ status: 'progress', progress: 100, file })
157
+ return hit.arrayBuffer()
158
+ }
159
+
160
+ const res = await fetch(url)
161
+ if (!res.ok) throw new Error(`Could not load ${file} (${res.status})`)
162
+
163
+ const total = Number(res.headers.get('content-length') ?? 0)
164
+ if (!res.body || !total) {
165
+ const buf = await res.arrayBuffer()
166
+ await cache?.put(url, new Response(buf.slice(0)))
167
+ return buf
168
+ }
169
+
170
+ const reader = res.body.getReader()
171
+ const chunks: Uint8Array[] = []
172
+ let received = 0
173
+ for (;;) {
174
+ const { done, value } = await reader.read()
175
+ if (done) break
176
+ chunks.push(value)
177
+ received += value.byteLength
178
+ onProgress?.({ status: 'progress', progress: (received / total) * 100, file })
179
+ }
180
+
181
+ const out = new Uint8Array(received)
182
+ let offset = 0
183
+ for (const chunk of chunks) {
184
+ out.set(chunk, offset)
185
+ offset += chunk.byteLength
186
+ }
187
+ await cache?.put(url, new Response(out.buffer.slice(0)))
188
+ return out.buffer
189
+ }
190
+
191
+ async function loadNpz(buffer: ArrayBuffer): Promise<Record<string, VoiceEntry>> {
192
+ const view = new DataView(buffer)
193
+ const bytes = new Uint8Array(buffer)
194
+ const voices: Record<string, VoiceEntry> = {}
195
+
196
+ const eocd = findEocd(view, bytes.length)
197
+ const count = view.getUint16(eocd + 10, true)
198
+ let central = view.getUint32(eocd + 16, true)
199
+
200
+ for (let i = 0; i < count; i++) {
201
+ if (view.getUint32(central, true) !== 0x02014b50) break
202
+ const method = view.getUint16(central + 10, true)
203
+ const compSize = view.getUint32(central + 20, true)
204
+ const rawSize = view.getUint32(central + 24, true)
205
+ const nameLen = view.getUint16(central + 28, true)
206
+ const extraLen = view.getUint16(central + 30, true)
207
+ const commentLen = view.getUint16(central + 32, true)
208
+ const localOff = view.getUint32(central + 42, true)
209
+ const name = new TextDecoder().decode(bytes.subarray(central + 46, central + 46 + nameLen))
210
+ central += 46 + nameLen + extraLen + commentLen
211
+ if (!name.endsWith('.npy')) continue
212
+
213
+ const localNameLen = view.getUint16(localOff + 26, true)
214
+ const localExtraLen = view.getUint16(localOff + 28, true)
215
+ const dataStart = localOff + 30 + localNameLen + localExtraLen
216
+ const compressed = bytes.subarray(dataStart, dataStart + compSize)
217
+ const raw =
218
+ method === 0 ? copyBytes(compressed) : method === 8 ? await inflateRaw(compressed, rawSize) : null
219
+ if (!raw) throw new Error(`Unsupported zip method ${method} in voices.npz`)
220
+ const key = name.replace(/\.npy$/, '').split('/').pop() ?? name
221
+ voices[key] = parseNpy(copyBytes(raw).buffer)
222
+ }
223
+
224
+ if (!Object.keys(voices).length) throw new Error('voices.npz contained no arrays')
225
+ return voices
226
+ }
227
+
228
+ function findEocd(view: DataView, length: number): number {
229
+ for (let i = length - 22; i >= Math.max(0, length - 22 - 0xffff); i--) {
230
+ if (view.getUint32(i, true) === 0x06054b50) return i
231
+ }
232
+ throw new Error('voices.npz is not a valid zip')
233
+ }
234
+
235
+ function copyBytes(src: Uint8Array): Uint8Array<ArrayBuffer> {
236
+ const out = new Uint8Array(new ArrayBuffer(src.byteLength))
237
+ out.set(src)
238
+ return out
239
+ }
240
+
241
+ async function inflateRaw(data: Uint8Array, outLen: number): Promise<Uint8Array> {
242
+ const stream = new DecompressionStream('deflate-raw')
243
+ const writer = stream.writable.getWriter()
244
+ await writer.write(Uint8Array.from(data))
245
+ await writer.close()
246
+ const reader = stream.readable.getReader()
247
+ const chunks: Uint8Array[] = []
248
+ let received = 0
249
+ for (;;) {
250
+ const { done, value } = await reader.read()
251
+ if (done) break
252
+ chunks.push(value)
253
+ received += value.byteLength
254
+ }
255
+ const out = new Uint8Array(outLen || received)
256
+ let offset = 0
257
+ for (const chunk of chunks) {
258
+ out.set(chunk, offset)
259
+ offset += chunk.byteLength
260
+ }
261
+ return out
262
+ }
263
+
264
+ function parseNpy(buf: ArrayBuffer): VoiceEntry {
265
+ const bytes = new Uint8Array(buf)
266
+ const major = bytes[6] ?? 1
267
+ const headerLen =
268
+ major >= 2 ? new DataView(buf, 8, 4).getUint32(0, true) : new DataView(buf, 8, 2).getUint16(0, true)
269
+ const headerOffset = major >= 2 ? 12 : 10
270
+ const header = new TextDecoder().decode(bytes.subarray(headerOffset, headerOffset + headerLen))
271
+ const shapeStr = header.match(/'shape'\s*:\s*\(([^)]*)\)/)?.[1]?.trim() ?? ''
272
+ const shape = shapeStr === '' ? [1] : shapeStr.split(',').map((s) => Number(s.trim())).filter((n) => !Number.isNaN(n))
273
+ const start = headerOffset + headerLen
274
+ const avail = buf.byteLength - start
275
+ const usable = avail - (avail % 4)
276
+ const data = new Float32Array(buf.slice(start, start + usable))
277
+ return { data, shape }
278
+ }
279
+
280
+ export { SAMPLE_RATE as KITTEN_SAMPLE_RATE }
@@ -0,0 +1,140 @@
1
+ import type { Device, ProgressCallback, ResolvedDevice } from '../core/index.ts'
2
+ import { resolveDevice, toProgressInfo } from '../core/index.ts'
3
+ import { phonemizeLang } from './phonemes.ts'
4
+ import { splitUtterances } from './split.ts'
5
+ import { toFloat32 } from './wav.ts'
6
+ import { kokoroPhonemeLang } from './voices.ts'
7
+ import type { SpeakChunk } from './types.ts'
8
+
9
+ /**
10
+ * Kokoro 82M (StyleTTS 2) on Transformers.js directly. No `kokoro-js`, so the app ships one
11
+ * copy of Transformers.js. Voice style vectors come from the same Hugging Face repo.
12
+ */
13
+ const DEFAULT_MODEL = 'onnx-community/Kokoro-82M-v1.0-ONNX'
14
+ const SAMPLE_RATE = 24_000
15
+ const STYLE_DIM = 256
16
+ const MAX_STYLE_INDEX = 509
17
+ const VOICE_CACHE = 'runonweb-kokoro-voices'
18
+
19
+ type Tensorish = { data: Float32Array | number[]; dims: number[] }
20
+
21
+ type KokoroModel = {
22
+ (inputs: Record<string, unknown>): Promise<{ waveform: Tensorish }>
23
+ dispose?: () => Promise<void>
24
+ }
25
+
26
+ type KokoroTokenizer = (text: string, options?: { truncation?: boolean }) => { input_ids: { dims: number[] } }
27
+
28
+ export type LoadedKokoro = {
29
+ model: KokoroModel
30
+ tokenizer: KokoroTokenizer
31
+ modelId: string
32
+ device: ResolvedDevice
33
+ }
34
+
35
+ export async function loadKokoro(options: {
36
+ model?: string
37
+ device?: Device
38
+ onProgress?: ProgressCallback
39
+ }): Promise<LoadedKokoro> {
40
+ const device = await resolveDevice(options.device ?? 'auto')
41
+ options.onProgress?.({ status: 'loading', progress: 0 })
42
+
43
+ const { StyleTextToSpeech2Model, AutoTokenizer, env } = await import('@huggingface/transformers')
44
+ env.allowLocalModels = false
45
+ const modelId = options.model ?? DEFAULT_MODEL
46
+ const progress_callback = (data: Record<string, unknown>) => {
47
+ options.onProgress?.(toProgressInfo(data))
48
+ }
49
+
50
+ const [model, tokenizer] = await Promise.all([
51
+ StyleTextToSpeech2Model.from_pretrained(modelId, {
52
+ device,
53
+ // fp16 / q4f16 produce NaN on the Transformers.js 4 WebGPU runtime; fp32 is clean and fast.
54
+ dtype: device === 'webgpu' ? 'fp32' : 'q8',
55
+ progress_callback,
56
+ }),
57
+ AutoTokenizer.from_pretrained(modelId, { progress_callback }),
58
+ ])
59
+
60
+ options.onProgress?.({ status: 'ready', progress: 100 })
61
+ return {
62
+ model: model as unknown as KokoroModel,
63
+ tokenizer: tokenizer as unknown as KokoroTokenizer,
64
+ modelId,
65
+ device,
66
+ }
67
+ }
68
+
69
+ export async function* streamKokoro(
70
+ kokoro: LoadedKokoro,
71
+ text: string,
72
+ voice: string,
73
+ speed: number
74
+ ): AsyncGenerator<SpeakChunk> {
75
+ const lang = kokoroPhonemeLang(voice) ?? (voice.startsWith('b') ? 'en-gb' : 'en-us')
76
+ for (const piece of splitUtterances(text)) {
77
+ const phonemes = await phonemizeLang(piece, lang)
78
+ if (!phonemes) throw new Error(`Could not phonemize text for ${lang}`)
79
+ const audio = await synthesize(kokoro, phonemes, voice, speed)
80
+ yield { audio, samplingRate: SAMPLE_RATE, text: piece }
81
+ }
82
+ }
83
+
84
+ async function synthesize(kokoro: LoadedKokoro, phonemes: string, voice: string, speed: number): Promise<Float32Array> {
85
+ const { Tensor } = await import('@huggingface/transformers')
86
+ const { input_ids } = kokoro.tokenizer(phonemes, { truncation: true })
87
+ const numTokens = Math.min(Math.max((input_ids.dims.at(-1) ?? 2) - 2, 0), MAX_STYLE_INDEX)
88
+ const styles = await loadVoice(kokoro.modelId, voice)
89
+ const style = styles.slice(numTokens * STYLE_DIM, (numTokens + 1) * STYLE_DIM)
90
+
91
+ const { waveform } = await kokoro.model({
92
+ input_ids,
93
+ style: new Tensor('float32', style, [1, STYLE_DIM]),
94
+ speed: new Tensor('float32', [speed], [1]),
95
+ })
96
+ return toFloat32(waveform.data)
97
+ }
98
+
99
+ const voiceCache = new Map<string, Promise<Float32Array>>()
100
+
101
+ /** Fetch `voices/<id>.bin` (510×256 float32 styles indexed by token count) and keep it in the Cache API. */
102
+ function loadVoice(modelId: string, voice: string): Promise<Float32Array> {
103
+ const key = `${modelId}/${voice}`
104
+ let pending = voiceCache.get(key)
105
+ if (!pending) {
106
+ pending = fetchVoice(modelId, voice).catch((err) => {
107
+ voiceCache.delete(key)
108
+ throw err
109
+ })
110
+ voiceCache.set(key, pending)
111
+ }
112
+ return pending
113
+ }
114
+
115
+ async function fetchVoice(modelId: string, voice: string): Promise<Float32Array> {
116
+ const url = `https://huggingface.co/${modelId}/resolve/main/voices/${voice}.bin`
117
+ let cache: Cache | null = null
118
+ try {
119
+ cache = await caches.open(VOICE_CACHE)
120
+ const hit = await cache.match(url)
121
+ if (hit) return new Float32Array(await hit.arrayBuffer())
122
+ } catch {
123
+ cache = null
124
+ }
125
+ const res = await fetch(url)
126
+ if (!res.ok) throw new Error(`Voice "${voice}" not found (${res.status})`)
127
+ const buffer = await res.arrayBuffer()
128
+ try {
129
+ await cache?.put(url, new Response(buffer.slice(0), { headers: { 'content-type': 'application/octet-stream' } }))
130
+ } catch {
131
+ // Cache is best-effort.
132
+ }
133
+ return new Float32Array(buffer)
134
+ }
135
+
136
+ export function disposeKokoro(kokoro: LoadedKokoro | null) {
137
+ void kokoro?.model.dispose?.()
138
+ }
139
+
140
+ export { DEFAULT_MODEL as KOKORO_MODEL, SAMPLE_RATE as KOKORO_SAMPLE_RATE }
@@ -0,0 +1,150 @@
1
+ import { kokoroPhonemeLang } from './voices.ts'
2
+
3
+ const ESPEAK_JS = 'https://cdn.jsdelivr.net/npm/espeak-ng@1.0.2/dist/espeak-ng.js'
4
+ const ESPEAK_WASM = 'https://cdn.jsdelivr.net/npm/espeak-ng@1.0.2/dist/espeak-ng.wasm'
5
+
6
+ type ESpeakInstance = {
7
+ FS: { readFile: (name: string, opts: { encoding: string }) => string }
8
+ }
9
+
10
+ type ESpeakFactory = (opts: {
11
+ locateFile?: (file: string) => string
12
+ arguments?: string[]
13
+ }) => Promise<ESpeakInstance>
14
+
15
+ let factory: ESpeakFactory | null = null
16
+
17
+ /**
18
+ * Convert text to IPA for Kokoro. English uses `phonemizer` with Kokoro's text normalization
19
+ * and phoneme fixes (ported from kokoro-js); ES/FR load eSpeak-NG WASM on demand.
20
+ */
21
+ export async function phonemizeLang(text: string, lang: string): Promise<string> {
22
+ if (lang === 'en-us' || lang === 'en-gb' || lang === 'en') {
23
+ return phonemizeEnglish(text, lang === 'en-us' ? 'a' : 'b')
24
+ }
25
+ return phonemizeEspeak(text, lang)
26
+ }
27
+
28
+ const PUNCT = ';:,.!?¡¿—…"«»“”(){}[]'
29
+ const PUNCT_RE = new RegExp(`(\\s*[${PUNCT.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}]+\\s*)+`, 'g')
30
+
31
+ async function phonemizeEnglish(text: string, variant: 'a' | 'b'): Promise<string> {
32
+ const { phonemize } = await import('phonemizer')
33
+ const normalized = normalizeEnglish(text)
34
+ const lang = variant === 'a' ? 'en-us' : 'en'
35
+
36
+ // Phonemize the text between punctuation runs; keep the punctuation verbatim.
37
+ const parts: string[] = []
38
+ let last = 0
39
+ for (const m of normalized.matchAll(PUNCT_RE)) {
40
+ const idx = m.index ?? 0
41
+ if (last < idx) parts.push((await phonemize(normalized.slice(last, idx), lang)).join(' '))
42
+ if (m[0].length > 0) parts.push(m[0])
43
+ last = idx + m[0].length
44
+ }
45
+ if (last < normalized.length) parts.push((await phonemize(normalized.slice(last), lang)).join(' '))
46
+
47
+ let ps = parts
48
+ .join('')
49
+ .replace(/kəkˈoːɹoʊ/g, 'kˈoʊkəɹoʊ')
50
+ .replace(/kəkˈɔːɹəʊ/g, 'kˈəʊkəɹəʊ')
51
+ .replace(/ʲ/g, 'j')
52
+ .replace(/r/g, 'ɹ')
53
+ .replace(/x/g, 'k')
54
+ .replace(/ɬ/g, 'l')
55
+ .replace(/(?<=[a-zɹː])(?=hˈʌndɹɪd)/g, ' ')
56
+ .replace(/ z(?=[;:,.!?¡¿—…"«»“” ]|$)/g, 'z')
57
+ if (variant === 'a') ps = ps.replace(/(?<=nˈaɪn)ti(?!ː)/g, 'di')
58
+ return ps.replace(/\s+/g, ' ').trim()
59
+ }
60
+
61
+ function normalizeEnglish(text: string): string {
62
+ return text
63
+ .replace(/[‘’]/g, "'")
64
+ .replace(/«/g, '“')
65
+ .replace(/»/g, '”')
66
+ .replace(/[“”]/g, '"')
67
+ .replace(/\(/g, '«')
68
+ .replace(/\)/g, '»')
69
+ .replace(/、/g, ', ')
70
+ .replace(/。/g, '. ')
71
+ .replace(/!/g, '! ')
72
+ .replace(/,/g, ', ')
73
+ .replace(/:/g, ': ')
74
+ .replace(/;/g, '; ')
75
+ .replace(/?/g, '? ')
76
+ .replace(/[^\S \n]/g, ' ')
77
+ .replace(/ +/, ' ')
78
+ .replace(/(?<=\n) +(?=\n)/g, '')
79
+ .replace(/\bD[Rr]\.(?= [A-Z])/g, 'Doctor')
80
+ .replace(/\b(?:Mr\.|MR\.(?= [A-Z]))/g, 'Mister')
81
+ .replace(/\b(?:Ms\.|MS\.(?= [A-Z]))/g, 'Miss')
82
+ .replace(/\b(?:Mrs\.|MRS\.(?= [A-Z]))/g, 'Mrs')
83
+ .replace(/\betc\.(?! [A-Z])/gi, 'etc')
84
+ .replace(/\b(y)eah?\b/gi, "$1e'a")
85
+ .replace(/\d*\.\d+|\b\d{4}s?\b|(?<!:)\b(?:[1-9]|1[0-2]):[0-5]\d\b(?!:)/g, splitNum)
86
+ .replace(/(?<=\d),(?=\d)/g, '')
87
+ .replace(/[$£]\d+(?:\.\d+)?(?: hundred| thousand| (?:[bm]|tr)illion)*\b|[$£]\d+\.\d\d?\b/gi, flipMoney)
88
+ .replace(/\d*\.\d+/g, pointNum)
89
+ .replace(/(?<=\d)-(?=\d)/g, ' to ')
90
+ .replace(/(?<=\d)S/g, ' S')
91
+ .replace(/(?<=[BCDFGHJ-NP-TV-Z])'?s\b/g, "'S")
92
+ .replace(/(?<=X')S\b/g, 's')
93
+ .replace(/(?:[A-Za-z]\.){2,} [a-z]/g, (m) => m.replace(/\./g, '-'))
94
+ .replace(/(?<=[A-Z])\.(?=[A-Z])/gi, '-')
95
+ .trim()
96
+ }
97
+
98
+ function splitNum(match: string): string {
99
+ if (match.includes('.')) return match
100
+ if (match.includes(':')) {
101
+ const [h, m] = match.split(':').map(Number)
102
+ if (m === 0) return `${h} o'clock`
103
+ if (m! < 10) return `${h} oh ${m}`
104
+ return `${h} ${m}`
105
+ }
106
+ const year = parseInt(match.slice(0, 4), 10)
107
+ if (year < 1100 || year % 1000 < 10) return match
108
+ const left = match.slice(0, 2)
109
+ const right = parseInt(match.slice(2, 4), 10)
110
+ const suffix = match.endsWith('s') ? 's' : ''
111
+ if (year % 1000 >= 100 && year % 1000 <= 999) {
112
+ if (right === 0) return `${left} hundred${suffix}`
113
+ if (right < 10) return `${left} oh ${right}${suffix}`
114
+ }
115
+ return `${left} ${right}${suffix}`
116
+ }
117
+
118
+ function flipMoney(match: string): string {
119
+ const unit = match[0] === '$' ? 'dollar' : 'pound'
120
+ if (isNaN(Number(match.slice(1)))) return `${match.slice(1)} ${unit}s`
121
+ if (!match.includes('.')) {
122
+ const s = match.slice(1) === '1' ? '' : 's'
123
+ return `${match.slice(1)} ${unit}${s}`
124
+ }
125
+ const [whole, frac] = match.slice(1).split('.')
126
+ const cents = parseInt((frac ?? '').padEnd(2, '0'), 10)
127
+ const centUnit = match[0] === '$' ? (cents === 1 ? 'cent' : 'cents') : cents === 1 ? 'penny' : 'pence'
128
+ return `${whole} ${unit}${whole === '1' ? '' : 's'} and ${cents} ${centUnit}`
129
+ }
130
+
131
+ function pointNum(match: string): string {
132
+ const [whole, frac] = match.split('.')
133
+ return `${whole} point ${(frac ?? '').split('').join(' ')}`
134
+ }
135
+
136
+ async function phonemizeEspeak(text: string, lang: string): Promise<string> {
137
+ if (!factory) {
138
+ const mod = (await import(/* @vite-ignore */ ESPEAK_JS)) as { default: ESpeakFactory }
139
+ factory = mod.default
140
+ }
141
+ const instance = await factory({
142
+ locateFile: (file) => (file.endsWith('.wasm') ? ESPEAK_WASM : file),
143
+ arguments: ['--phonout', 'generated', '-q', '-b=1', '--ipa=3', '-v', lang, text],
144
+ })
145
+ return instance.FS.readFile('generated', { encoding: 'utf8' }).replace(/\s+/g, ' ').trim()
146
+ }
147
+
148
+ export function needsExternalPhonemes(voiceId: string): boolean {
149
+ return kokoroPhonemeLang(voiceId) != null
150
+ }
@@ -0,0 +1,30 @@
1
+ export type TTSSize = 'tiny' | 'small' | 'multi'
2
+
3
+ export const TTS_SIZES: Record<
4
+ TTSSize,
5
+ { label: string; engine: string; params: string; downloadMB: string; quality: string }
6
+ > = {
7
+ tiny: {
8
+ label: 'Tiny',
9
+ engine: 'KittenTTS',
10
+ params: '15M',
11
+ downloadMB: '~28 MB',
12
+ quality: '8 English voices',
13
+ },
14
+ small: {
15
+ label: 'Small',
16
+ engine: 'Kokoro 82M',
17
+ params: '82M',
18
+ downloadMB: '~326 MB · WASM ~92 MB',
19
+ quality: 'English · Spanish · French',
20
+ },
21
+ multi: {
22
+ label: 'Multi',
23
+ engine: 'Supertonic 2',
24
+ params: '66M',
25
+ downloadMB: '~262 MB',
26
+ quality: 'en · ko · es · pt · fr · 44.1 kHz',
27
+ },
28
+ }
29
+
30
+ export const DEFAULT_TTS_SIZE: TTSSize = 'small'
@@ -0,0 +1,31 @@
1
+ const MAX_CHARS = 240
2
+
3
+ /** Split text into sentence-sized chunks for streaming synthesis. */
4
+ export function splitUtterances(text: string): string[] {
5
+ const trimmed = text.trim()
6
+ if (!trimmed) return []
7
+
8
+ const sentences = trimmed.split(/(?<=[.!?…])\s+|\n+/).map((s) => s.trim()).filter(Boolean)
9
+ const out: string[] = []
10
+
11
+ for (const sentence of sentences) {
12
+ if (sentence.length <= MAX_CHARS) {
13
+ out.push(sentence)
14
+ continue
15
+ }
16
+ const words = sentence.split(/\s+/)
17
+ let buf = ''
18
+ for (const word of words) {
19
+ const next = buf ? `${buf} ${word}` : word
20
+ if (next.length > MAX_CHARS && buf) {
21
+ out.push(buf)
22
+ buf = word
23
+ } else {
24
+ buf = next
25
+ }
26
+ }
27
+ if (buf) out.push(buf)
28
+ }
29
+
30
+ return out
31
+ }