use-voice-control 0.1.95 → 0.1.97
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli-CXzTD4W4.js +338 -0
- package/dist/cli-CXzTD4W4.js.map +1 -0
- package/dist/cli.js +2 -374
- package/dist/client-CwHmea9T.js +229 -0
- package/dist/client-CwHmea9T.js.map +1 -0
- package/dist/client.js +3 -276
- package/dist/index.js +45 -55
- package/dist/index.js.map +1 -1
- package/dist/kokoro-node-CDhCeD6u.js +115 -0
- package/dist/kokoro-node-CDhCeD6u.js.map +1 -0
- package/dist/markdown.js +179 -206
- package/dist/markdown.js.map +1 -1
- package/dist/node.js +4 -27
- package/dist/react.js +188 -211
- package/dist/react.js.map +1 -1
- package/dist/semantic-split-DF26uo-Z.js +36 -0
- package/dist/semantic-split-DF26uo-Z.js.map +1 -0
- package/package.json +14 -14
- package/dist/cli.js.map +0 -1
- package/dist/client.js.map +0 -1
- package/dist/kokoro-node-DJ_Rxp_N.js +0 -144
- package/dist/kokoro-node-DJ_Rxp_N.js.map +0 -1
- package/dist/node.js.map +0 -1
- package/dist/semantic-split-CXhk-k1F.js +0 -44
- package/dist/semantic-split-CXhk-k1F.js.map +0 -1
package/dist/client.js
CHANGED
|
@@ -1,276 +1,3 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
import { looksLikeMarkdown as E, markdownToSpeech as L } from "./markdown.js";
|
|
5
|
-
import { markdownToSpeechSegments as D, stripInlineMarkdown as N } from "./markdown.js";
|
|
6
|
-
import { s as C } from "./semantic-split-CXhk-k1F.js";
|
|
7
|
-
const T = "/api/speech/tts", v = "af_heart", R = 240;
|
|
8
|
-
function w(r) {
|
|
9
|
-
return r.aborted ? Promise.resolve() : new Promise((t) => {
|
|
10
|
-
r.addEventListener("abort", () => t(), { once: !0 });
|
|
11
|
-
});
|
|
12
|
-
}
|
|
13
|
-
class O {
|
|
14
|
-
constructor(t = {}) {
|
|
15
|
-
a(this, "options");
|
|
16
|
-
a(this, "state", "idle");
|
|
17
|
-
a(this, "abortController", null);
|
|
18
|
-
a(this, "audio", null);
|
|
19
|
-
a(this, "objectUrl", null);
|
|
20
|
-
/** Set once the endpoint has proved unreachable, so later chunks skip the retry. */
|
|
21
|
-
a(this, "endpointUnavailable", !1);
|
|
22
|
-
this.options = t;
|
|
23
|
-
}
|
|
24
|
-
/** Replace the options (voice, callbacks, …) without discarding playback state. */
|
|
25
|
-
setOptions(t) {
|
|
26
|
-
this.options = t;
|
|
27
|
-
}
|
|
28
|
-
getState() {
|
|
29
|
-
return this.state;
|
|
30
|
-
}
|
|
31
|
-
isActive() {
|
|
32
|
-
return this.state === "speaking" || this.state === "paused" || this.state === "loading";
|
|
33
|
-
}
|
|
34
|
-
/**
|
|
35
|
-
* Speak `text`, cancelling anything already playing. Resolves when playback
|
|
36
|
-
* finishes or is stopped — it never rejects; failures go to `onError`.
|
|
37
|
-
*/
|
|
38
|
-
async speak(t) {
|
|
39
|
-
var o, h, c, f, u, g, d, y;
|
|
40
|
-
this.stop();
|
|
41
|
-
const e = this.options.maxChunkLength ?? R, i = C(this.toSpeakableText(t ?? ""), e).map((l) => l.trim()).filter((l) => l.length > 0);
|
|
42
|
-
if (i.length === 0) return;
|
|
43
|
-
const s = new AbortController();
|
|
44
|
-
this.abortController = s;
|
|
45
|
-
const { signal: n } = s;
|
|
46
|
-
this.setState("loading");
|
|
47
|
-
try {
|
|
48
|
-
let l = this.synthesize(i[0], n);
|
|
49
|
-
for (let p = 0; p < i.length; p += 1) {
|
|
50
|
-
const S = l;
|
|
51
|
-
l = p + 1 < i.length ? this.synthesize(i[p + 1], n) : Promise.resolve(null);
|
|
52
|
-
const b = await S;
|
|
53
|
-
if (n.aborted || (this.setState("speaking"), (h = (o = this.options).onChunk) == null || h.call(o, { text: i[p], index: p, total: i.length }), n.aborted) || (await this.playChunk(b, i[p], n), n.aborted)) return;
|
|
54
|
-
}
|
|
55
|
-
this.cleanup(), this.setState("idle"), (f = (c = this.options).onEnd) == null || f.call(c, "finished");
|
|
56
|
-
} catch (l) {
|
|
57
|
-
if (n.aborted) return;
|
|
58
|
-
this.cleanup(), this.setState("idle"), (g = (u = this.options).onError) == null || g.call(
|
|
59
|
-
u,
|
|
60
|
-
l instanceof Error ? l : new Error(String(l))
|
|
61
|
-
), (y = (d = this.options).onEnd) == null || y.call(d, "stopped");
|
|
62
|
-
} finally {
|
|
63
|
-
this.abortController === s && (this.abortController = null);
|
|
64
|
-
}
|
|
65
|
-
}
|
|
66
|
-
pause() {
|
|
67
|
-
this.state === "speaking" && (this.audio ? this.audio.pause() : typeof speechSynthesis < "u" && speechSynthesis.pause(), this.setState("paused"));
|
|
68
|
-
}
|
|
69
|
-
resume() {
|
|
70
|
-
this.state === "paused" && (this.audio ? this.audio.play().catch(() => {
|
|
71
|
-
}) : typeof speechSynthesis < "u" && speechSynthesis.resume(), this.setState("speaking"));
|
|
72
|
-
}
|
|
73
|
-
/** Stop playback and drop any queued chunks. Safe to call when already idle. */
|
|
74
|
-
stop() {
|
|
75
|
-
var e, i, s;
|
|
76
|
-
const t = this.isActive();
|
|
77
|
-
(e = this.abortController) == null || e.abort(), this.abortController = null, this.cleanup(), t && (this.setState("idle"), (s = (i = this.options).onEnd) == null || s.call(i, "stopped"));
|
|
78
|
-
}
|
|
79
|
-
/**
|
|
80
|
-
* Converts Markdown to spoken words before chunking, so a document read out
|
|
81
|
-
* of an editor does not have its syntax read back to the listener.
|
|
82
|
-
*/
|
|
83
|
-
toSpeakableText(t) {
|
|
84
|
-
const e = this.options.format ?? "auto";
|
|
85
|
-
return e === "text" ? t : e === "markdown" || E(t) ? L(t, this.options.markdown) : t;
|
|
86
|
-
}
|
|
87
|
-
setState(t) {
|
|
88
|
-
var e, i;
|
|
89
|
-
this.state !== t && (this.state = t, (i = (e = this.options).onStateChange) == null || i.call(e, t));
|
|
90
|
-
}
|
|
91
|
-
cleanup() {
|
|
92
|
-
this.audio && (this.audio.pause(), this.audio.src = "", this.audio = null), this.objectUrl && (URL.revokeObjectURL(this.objectUrl), this.objectUrl = null), typeof speechSynthesis < "u" && speechSynthesis.cancel();
|
|
93
|
-
}
|
|
94
|
-
synthesize(t, e) {
|
|
95
|
-
const i = this.options.synthesize ? this.options.synthesize(t, e) : this.fetchAudio(t, e);
|
|
96
|
-
return i.catch(() => {
|
|
97
|
-
}), i;
|
|
98
|
-
}
|
|
99
|
-
async fetchAudio(t, e) {
|
|
100
|
-
if (this.endpointUnavailable) return null;
|
|
101
|
-
const i = this.options.endpoint ?? T;
|
|
102
|
-
try {
|
|
103
|
-
const s = await fetch(i, {
|
|
104
|
-
method: "POST",
|
|
105
|
-
headers: { "Content-Type": "application/json" },
|
|
106
|
-
body: JSON.stringify({
|
|
107
|
-
text: t,
|
|
108
|
-
provider: this.options.provider ?? "kokoro",
|
|
109
|
-
voice: this.options.voice ?? v
|
|
110
|
-
}),
|
|
111
|
-
signal: e
|
|
112
|
-
});
|
|
113
|
-
if (!s.ok) throw new Error(`TTS failed: ${s.status}`);
|
|
114
|
-
const n = s.headers.get("Content-Type") || "audio/wav";
|
|
115
|
-
return new Blob([await s.arrayBuffer()], { type: n });
|
|
116
|
-
} catch (s) {
|
|
117
|
-
if (e.aborted) throw s;
|
|
118
|
-
return this.endpointUnavailable = !0, null;
|
|
119
|
-
}
|
|
120
|
-
}
|
|
121
|
-
playChunk(t, e, i) {
|
|
122
|
-
return t ? this.playAudioBlob(t, i) : this.playWithSpeechSynthesis(e, i);
|
|
123
|
-
}
|
|
124
|
-
async playAudioBlob(t, e) {
|
|
125
|
-
const i = URL.createObjectURL(t), s = new Audio(i);
|
|
126
|
-
this.audio = s, this.objectUrl = i;
|
|
127
|
-
const n = new Promise((o, h) => {
|
|
128
|
-
s.addEventListener("ended", () => o(), { once: !0 }), s.addEventListener(
|
|
129
|
-
"error",
|
|
130
|
-
() => h(new Error("Audio playback failed")),
|
|
131
|
-
{ once: !0 }
|
|
132
|
-
);
|
|
133
|
-
});
|
|
134
|
-
try {
|
|
135
|
-
await s.play(), await Promise.race([n, w(e)]);
|
|
136
|
-
} finally {
|
|
137
|
-
this.audio === s && (this.audio = null), this.objectUrl === i && (this.objectUrl = null), s.pause(), URL.revokeObjectURL(i);
|
|
138
|
-
}
|
|
139
|
-
}
|
|
140
|
-
async playWithSpeechSynthesis(t, e) {
|
|
141
|
-
if (typeof speechSynthesis > "u" || typeof SpeechSynthesisUtterance > "u")
|
|
142
|
-
throw new Error("No speech synthesis available in this browser");
|
|
143
|
-
const i = new SpeechSynthesisUtterance(t), s = new Promise((n, o) => {
|
|
144
|
-
i.onend = () => n(), i.onerror = () => e.aborted ? n() : o(new Error("Speech synthesis failed"));
|
|
145
|
-
});
|
|
146
|
-
speechSynthesis.speak(i), await Promise.race([s, w(e)]);
|
|
147
|
-
}
|
|
148
|
-
}
|
|
149
|
-
function m() {
|
|
150
|
-
return typeof window > "u" ? null : window.SpeechRecognition || window.webkitSpeechRecognition || null;
|
|
151
|
-
}
|
|
152
|
-
function x() {
|
|
153
|
-
var r;
|
|
154
|
-
return typeof window > "u" ? !1 : m() ? !0 : !!((r = navigator.mediaDevices) != null && r.getUserMedia);
|
|
155
|
-
}
|
|
156
|
-
class M {
|
|
157
|
-
constructor(t = {}) {
|
|
158
|
-
a(this, "options");
|
|
159
|
-
a(this, "listening", !1);
|
|
160
|
-
/** Distinguishes a deliberate `stop()` from Chromium's idle auto-stop. */
|
|
161
|
-
a(this, "stopRequested", !1);
|
|
162
|
-
a(this, "recognition", null);
|
|
163
|
-
a(this, "moonshine", null);
|
|
164
|
-
this.options = t;
|
|
165
|
-
}
|
|
166
|
-
setOptions(t) {
|
|
167
|
-
this.options = t;
|
|
168
|
-
}
|
|
169
|
-
isListening() {
|
|
170
|
-
return this.listening;
|
|
171
|
-
}
|
|
172
|
-
async start() {
|
|
173
|
-
var i, s;
|
|
174
|
-
if (this.listening) return;
|
|
175
|
-
this.stopRequested = !1;
|
|
176
|
-
const t = this.options.engine ?? "auto", e = m();
|
|
177
|
-
try {
|
|
178
|
-
if (t !== "moonshine" && e) {
|
|
179
|
-
this.startWebSpeech(e);
|
|
180
|
-
return;
|
|
181
|
-
}
|
|
182
|
-
if (t === "webspeech")
|
|
183
|
-
throw new Error("Speech recognition is not available in this browser");
|
|
184
|
-
await this.startMoonshine();
|
|
185
|
-
} catch (n) {
|
|
186
|
-
this.setListening(!1), (s = (i = this.options).onError) == null || s.call(
|
|
187
|
-
i,
|
|
188
|
-
n instanceof Error ? n : new Error(String(n))
|
|
189
|
-
);
|
|
190
|
-
}
|
|
191
|
-
}
|
|
192
|
-
async stop() {
|
|
193
|
-
var t, e;
|
|
194
|
-
if (this.stopRequested = !0, this.recognition) {
|
|
195
|
-
try {
|
|
196
|
-
this.recognition.stop();
|
|
197
|
-
} catch {
|
|
198
|
-
}
|
|
199
|
-
this.recognition = null;
|
|
200
|
-
}
|
|
201
|
-
if (this.moonshine) {
|
|
202
|
-
try {
|
|
203
|
-
await ((e = (t = this.moonshine).stop) == null ? void 0 : e.call(t));
|
|
204
|
-
} catch {
|
|
205
|
-
}
|
|
206
|
-
this.moonshine = null;
|
|
207
|
-
}
|
|
208
|
-
this.setListening(!1);
|
|
209
|
-
}
|
|
210
|
-
async toggle() {
|
|
211
|
-
this.listening ? await this.stop() : await this.start();
|
|
212
|
-
}
|
|
213
|
-
setListening(t) {
|
|
214
|
-
var e, i;
|
|
215
|
-
this.listening !== t && (this.listening = t, (i = (e = this.options).onStateChange) == null || i.call(e, t));
|
|
216
|
-
}
|
|
217
|
-
startWebSpeech(t) {
|
|
218
|
-
const e = new t();
|
|
219
|
-
e.continuous = !0, e.interimResults = !0, e.lang = this.options.language ?? "en-US", e.onstart = () => this.setListening(!0), e.onresult = (i) => {
|
|
220
|
-
var n, o, h, c, f;
|
|
221
|
-
let s = "";
|
|
222
|
-
for (let u = i.resultIndex; u < i.results.length; u += 1) {
|
|
223
|
-
const g = i.results[u], d = String(((n = g[0]) == null ? void 0 : n.transcript) ?? "").trim();
|
|
224
|
-
d && (g.isFinal ? (h = (o = this.options).onCommit) == null || h.call(o, d) : s += `${s ? " " : ""}${d}`);
|
|
225
|
-
}
|
|
226
|
-
s && ((f = (c = this.options).onPartial) == null || f.call(c, s));
|
|
227
|
-
}, e.onerror = (i) => {
|
|
228
|
-
var n, o;
|
|
229
|
-
const s = i == null ? void 0 : i.error;
|
|
230
|
-
s === "no-speech" || s === "aborted" || (this.stopRequested = !0, (o = (n = this.options).onError) == null || o.call(n, new Error(`Speech recognition error: ${s}`)));
|
|
231
|
-
}, e.onend = () => {
|
|
232
|
-
if (this.stopRequested || this.recognition !== e) {
|
|
233
|
-
this.recognition = null, this.setListening(!1);
|
|
234
|
-
return;
|
|
235
|
-
}
|
|
236
|
-
try {
|
|
237
|
-
e.start();
|
|
238
|
-
} catch {
|
|
239
|
-
this.recognition = null, this.setListening(!1);
|
|
240
|
-
}
|
|
241
|
-
}, this.recognition = e, e.start();
|
|
242
|
-
}
|
|
243
|
-
async startMoonshine() {
|
|
244
|
-
const t = await import("@moonshine-ai/moonshine-js"), e = new t.MicrophoneTranscriber(
|
|
245
|
-
this.options.model ?? "model/small",
|
|
246
|
-
{
|
|
247
|
-
onTranscriptionUpdated: (i) => {
|
|
248
|
-
var s, n;
|
|
249
|
-
(n = (s = this.options).onPartial) == null || n.call(s, String(i ?? "").trim());
|
|
250
|
-
},
|
|
251
|
-
onTranscriptionCommitted: (i) => {
|
|
252
|
-
var n, o, h, c;
|
|
253
|
-
const s = String(i ?? "").trim();
|
|
254
|
-
s ? (o = (n = this.options).onCommit) == null || o.call(n, s) : (c = (h = this.options).onPartial) == null || c.call(h, "");
|
|
255
|
-
}
|
|
256
|
-
},
|
|
257
|
-
!1
|
|
258
|
-
// streaming mode
|
|
259
|
-
);
|
|
260
|
-
if (this.moonshine = e, await e.start(), this.stopRequested) {
|
|
261
|
-
await this.stop();
|
|
262
|
-
return;
|
|
263
|
-
}
|
|
264
|
-
this.setListening(!0);
|
|
265
|
-
}
|
|
266
|
-
}
|
|
267
|
-
export {
|
|
268
|
-
M as LiveTranscriber,
|
|
269
|
-
O as ReadAloudController,
|
|
270
|
-
x as isTranscriptionSupported,
|
|
271
|
-
E as looksLikeMarkdown,
|
|
272
|
-
L as markdownToSpeech,
|
|
273
|
-
D as markdownToSpeechSegments,
|
|
274
|
-
N as stripInlineMarkdown
|
|
275
|
-
};
|
|
276
|
-
//# sourceMappingURL=client.js.map
|
|
1
|
+
import { looksLikeMarkdown as e, markdownToSpeech as t, markdownToSpeechSegments as n, stripInlineMarkdown as r } from "./markdown.js";
|
|
2
|
+
import { n as i, r as a, t as o } from "./client-CwHmea9T.js";
|
|
3
|
+
export { o as LiveTranscriber, a as ReadAloudController, i as isTranscriptionSupported, e as looksLikeMarkdown, t as markdownToSpeech, n as markdownToSpeechSegments, r as stripInlineMarkdown };
|
package/dist/index.js
CHANGED
|
@@ -1,59 +1,49 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
|
|
4
|
-
async function
|
|
5
|
-
|
|
6
|
-
|
|
1
|
+
import { a as e, c as t, i as n, l as r, o as i, s as a, u as o } from "./kokoro-node-CDhCeD6u.js";
|
|
2
|
+
import { looksLikeMarkdown as s, markdownToSpeech as c, markdownToSpeechSegments as l, stripInlineMarkdown as u } from "./markdown.js";
|
|
3
|
+
//#region speech/core/kokoro.ts
|
|
4
|
+
async function d(e, t = "af_heart", i = {}) {
|
|
5
|
+
let a = r.includes(t) ? t : "af_heart";
|
|
6
|
+
return n(e, {
|
|
7
|
+
...i,
|
|
8
|
+
voice: a
|
|
9
|
+
});
|
|
7
10
|
}
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
11
|
+
//#endregion
|
|
12
|
+
//#region speech/core/deepgram.ts
|
|
13
|
+
function f() {
|
|
14
|
+
let e = globalThis;
|
|
15
|
+
try {
|
|
16
|
+
if (typeof e.getCloudflareContext == "function") {
|
|
17
|
+
let t = e.getCloudflareContext();
|
|
18
|
+
if (t?.env?.AI) return t.env.AI;
|
|
19
|
+
}
|
|
20
|
+
} catch {}
|
|
21
|
+
return e.__env__?.AI ?? e.AI;
|
|
19
22
|
}
|
|
20
|
-
async function
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
};
|
|
23
|
+
async function p(e, n = "angus") {
|
|
24
|
+
let r = t.includes(n) ? n : "angus", i = f();
|
|
25
|
+
if (!i) throw Error("Cloudflare AI binding not available");
|
|
26
|
+
return {
|
|
27
|
+
audio: await i.run("@cf/deepgram/aura-1", {
|
|
28
|
+
text: e.slice(0, 5e3),
|
|
29
|
+
speaker: r,
|
|
30
|
+
encoding: "mp3"
|
|
31
|
+
}),
|
|
32
|
+
contentType: "audio/mpeg"
|
|
33
|
+
};
|
|
32
34
|
}
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
throw new Error(`Unknown TTS provider: ${n}`);
|
|
44
|
-
}
|
|
35
|
+
//#endregion
|
|
36
|
+
//#region speech/index.ts
|
|
37
|
+
async function m(e) {
|
|
38
|
+
let { text: t, provider: n = "kokoro", voice: r = "af_heart" } = e;
|
|
39
|
+
if (!t || typeof t != "string" || t.trim().length === 0) throw Error("Text is required");
|
|
40
|
+
switch (n) {
|
|
41
|
+
case "kokoro": return d(t, r);
|
|
42
|
+
case "deepgram": return p(t, r);
|
|
43
|
+
default: throw Error(`Unknown TTS provider: ${n}`);
|
|
44
|
+
}
|
|
45
45
|
}
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
k as describeKokoroVoice,
|
|
51
|
-
w as encodeWav,
|
|
52
|
-
l as generateSpeech,
|
|
53
|
-
v as looksLikeMarkdown,
|
|
54
|
-
_ as markdownToSpeech,
|
|
55
|
-
y as markdownToSpeechSegments,
|
|
56
|
-
A as stripInlineMarkdown,
|
|
57
|
-
m as wavDurationSeconds
|
|
58
|
-
};
|
|
59
|
-
//# sourceMappingURL=index.js.map
|
|
46
|
+
//#endregion
|
|
47
|
+
export { t as DEEPGRAM_SPEAKERS, r as KOKORO_VOICES, e as concatSamples, o as describeKokoroVoice, i as encodeWav, m as generateSpeech, s as looksLikeMarkdown, c as markdownToSpeech, l as markdownToSpeechSegments, u as stripInlineMarkdown, a as wavDurationSeconds };
|
|
48
|
+
|
|
49
|
+
//# sourceMappingURL=index.js.map
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":["../speech/core/kokoro.ts","../speech/core/deepgram.ts","../speech/index.ts"],"sourcesContent":["/**\n * @fileoverview Kokoro TTS provider for `generateSpeech`.\n *\n * The work happens in `kokoro-node.ts`, which runs the Kokoro model locally\n * through `kokoro-js`; this file is the thin provider adapter that validates the\n * voice and returns the package's `TTSResult` shape.\n */\nimport type { TTSResult } from \"../types/types\";\nimport { KOKORO_VOICES, type KokoroVoice } from \"../types/types\";\nimport { synthesizeWav, type KokoroNodeOptions } from \"./kokoro-node\";\n\n/**\n * Generate speech from text using Kokoro.\n *\n * Long text is chunked and joined automatically. An unrecognised voice falls\n * back to `af_heart` rather than failing the request.\n */\nexport async function generateKokoroSpeech(\n text: string,\n voice: string = \"af_heart\",\n options: Omit<KokoroNodeOptions, \"voice\"> = {}\n): Promise<TTSResult> {\n const kokoroVoice = KOKORO_VOICES.includes(voice as KokoroVoice)\n ? (voice as KokoroVoice)\n : \"af_heart\";\n\n return synthesizeWav(text, { ...options, voice: kokoroVoice });\n}\n","/**\n * @fileoverview Deepgram TTS provider implementation using Cloudflare Workers AI\n * Requires Cloudflare AI binding\n */\nimport type { TTSResult } from \"../types/types\";\nimport { DEEPGRAM_SPEAKERS, type DeepgramSpeaker } from \"../types/types\";\n\n/**\n * Resolve the Cloudflare Workers AI binding at runtime without a hard\n * dependency on the host application. Consumers running on Cloudflare can\n * expose the context by setting `globalThis.getCloudflareContext` (as the\n * `@opennextjs/cloudflare` / `@cloudflare/next-on-pages` helpers do) or by\n * placing the bound `env` on `globalThis.__env__`.\n */\nfunction resolveCloudflareAI(): any {\n const g = globalThis as any;\n try {\n if (typeof g.getCloudflareContext === \"function\") {\n const ctx = g.getCloudflareContext();\n if (ctx?.env?.AI) return ctx.env.AI;\n }\n } catch {\n // CF context helper threw (e.g. called outside a request scope)\n }\n return g.__env__?.AI ?? g.AI;\n}\n\n/**\n * Generate speech from text using Deepgram Aura via Cloudflare Workers AI\n */\nexport async function generateDeepgramSpeech(\n text: string,\n speaker: string = \"angus\"\n): Promise<TTSResult> {\n // Validate speaker\n const auraVoice = DEEPGRAM_SPEAKERS.includes(speaker as DeepgramSpeaker)\n ? (speaker as DeepgramSpeaker)\n : \"angus\";\n\n const ai = resolveCloudflareAI();\n\n if (!ai) {\n throw new Error(\"Cloudflare AI binding not available\");\n }\n\n const result = await ai.run(\"@cf/deepgram/aura-1\", {\n text: text.slice(0, 5000),\n speaker: auraVoice,\n encoding: \"mp3\",\n });\n\n return {\n audio: result,\n contentType: \"audio/mpeg\",\n };\n}\n","/**\n * @fileoverview Unified text-to-speech API supporting Kokoro (default) and Deepgram\n *\n * Kokoro: Faster, more natural, runs on Node CPU\n * Deepgram: Requires Cloudflare AI binding, MP3 output\n */\nimport type { TTSOptions, TTSResult } from \"./types/types\";\nimport { generateKokoroSpeech } from \"./core/kokoro\";\nimport { generateDeepgramSpeech } from \"./core/deepgram\";\n\nexport * from \"./types/types\";\n\n// Markdown handling is exported from the root entry so a server route can turn a\n// document into speakable text with the same rules the CLI and the browser use.\nexport {\n looksLikeMarkdown,\n markdownToSpeech,\n markdownToSpeechSegments,\n stripInlineMarkdown,\n type MarkdownToSpeechOptions,\n type SpeechSegment,\n type SpeechSegmentType,\n} from \"./utils/markdown-to-speech\";\n\nexport { encodeWav, concatSamples, wavDurationSeconds } from \"./utils/wav\";\n\n/**\n * Generate speech from text using the specified provider\n *\n * @param options - TTS configuration\n * @returns Audio buffer and content type\n *\n * @example\n * ```ts\n * // Use Kokoro (default)\n * const audio = await generateSpeech({\n * text: \"Hello world\",\n * voice: \"af_heart\"\n * });\n *\n * // Use Deepgram\n * const audio = await generateSpeech({\n * text: \"Hello world\",\n * provider: \"deepgram\",\n * voice: \"angus\"\n * });\n * ```\n */\nexport async function generateSpeech(\n options: TTSOptions\n): Promise<TTSResult> {\n const { text, provider = \"kokoro\", voice = \"af_heart\" } = options;\n\n if (!text || typeof text !== \"string\" || text.trim().length === 0) {\n throw new Error(\"Text is required\");\n }\n\n switch (provider) {\n case \"kokoro\":\n return generateKokoroSpeech(text, voice);\n\n case \"deepgram\":\n return generateDeepgramSpeech(text, voice);\n\n default:\n throw new Error(`Unknown TTS provider: ${provider}`);\n }\n}\n"],"
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../speech/core/kokoro.ts","../speech/core/deepgram.ts","../speech/index.ts"],"sourcesContent":["/**\n * @fileoverview Kokoro TTS provider for `generateSpeech`.\n *\n * The work happens in `kokoro-node.ts`, which runs the Kokoro model locally\n * through `kokoro-js`; this file is the thin provider adapter that validates the\n * voice and returns the package's `TTSResult` shape.\n */\nimport type { TTSResult } from \"../types/types\";\nimport { KOKORO_VOICES, type KokoroVoice } from \"../types/types\";\nimport { synthesizeWav, type KokoroNodeOptions } from \"./kokoro-node\";\n\n/**\n * Generate speech from text using Kokoro.\n *\n * Long text is chunked and joined automatically. An unrecognised voice falls\n * back to `af_heart` rather than failing the request.\n */\nexport async function generateKokoroSpeech(\n text: string,\n voice: string = \"af_heart\",\n options: Omit<KokoroNodeOptions, \"voice\"> = {}\n): Promise<TTSResult> {\n const kokoroVoice = KOKORO_VOICES.includes(voice as KokoroVoice)\n ? (voice as KokoroVoice)\n : \"af_heart\";\n\n return synthesizeWav(text, { ...options, voice: kokoroVoice });\n}\n","/**\n * @fileoverview Deepgram TTS provider implementation using Cloudflare Workers AI\n * Requires Cloudflare AI binding\n */\nimport type { TTSResult } from \"../types/types\";\nimport { DEEPGRAM_SPEAKERS, type DeepgramSpeaker } from \"../types/types\";\n\n/**\n * Resolve the Cloudflare Workers AI binding at runtime without a hard\n * dependency on the host application. Consumers running on Cloudflare can\n * expose the context by setting `globalThis.getCloudflareContext` (as the\n * `@opennextjs/cloudflare` / `@cloudflare/next-on-pages` helpers do) or by\n * placing the bound `env` on `globalThis.__env__`.\n */\nfunction resolveCloudflareAI(): any {\n const g = globalThis as any;\n try {\n if (typeof g.getCloudflareContext === \"function\") {\n const ctx = g.getCloudflareContext();\n if (ctx?.env?.AI) return ctx.env.AI;\n }\n } catch {\n // CF context helper threw (e.g. called outside a request scope)\n }\n return g.__env__?.AI ?? g.AI;\n}\n\n/**\n * Generate speech from text using Deepgram Aura via Cloudflare Workers AI\n */\nexport async function generateDeepgramSpeech(\n text: string,\n speaker: string = \"angus\"\n): Promise<TTSResult> {\n // Validate speaker\n const auraVoice = DEEPGRAM_SPEAKERS.includes(speaker as DeepgramSpeaker)\n ? (speaker as DeepgramSpeaker)\n : \"angus\";\n\n const ai = resolveCloudflareAI();\n\n if (!ai) {\n throw new Error(\"Cloudflare AI binding not available\");\n }\n\n const result = await ai.run(\"@cf/deepgram/aura-1\", {\n text: text.slice(0, 5000),\n speaker: auraVoice,\n encoding: \"mp3\",\n });\n\n return {\n audio: result,\n contentType: \"audio/mpeg\",\n };\n}\n","/**\n * @fileoverview Unified text-to-speech API supporting Kokoro (default) and Deepgram\n *\n * Kokoro: Faster, more natural, runs on Node CPU\n * Deepgram: Requires Cloudflare AI binding, MP3 output\n */\nimport type { TTSOptions, TTSResult } from \"./types/types\";\nimport { generateKokoroSpeech } from \"./core/kokoro\";\nimport { generateDeepgramSpeech } from \"./core/deepgram\";\n\nexport * from \"./types/types\";\n\n// Markdown handling is exported from the root entry so a server route can turn a\n// document into speakable text with the same rules the CLI and the browser use.\nexport {\n looksLikeMarkdown,\n markdownToSpeech,\n markdownToSpeechSegments,\n stripInlineMarkdown,\n type MarkdownToSpeechOptions,\n type SpeechSegment,\n type SpeechSegmentType,\n} from \"./utils/markdown-to-speech\";\n\nexport { encodeWav, concatSamples, wavDurationSeconds } from \"./utils/wav\";\n\n/**\n * Generate speech from text using the specified provider\n *\n * @param options - TTS configuration\n * @returns Audio buffer and content type\n *\n * @example\n * ```ts\n * // Use Kokoro (default)\n * const audio = await generateSpeech({\n * text: \"Hello world\",\n * voice: \"af_heart\"\n * });\n *\n * // Use Deepgram\n * const audio = await generateSpeech({\n * text: \"Hello world\",\n * provider: \"deepgram\",\n * voice: \"angus\"\n * });\n * ```\n */\nexport async function generateSpeech(\n options: TTSOptions\n): Promise<TTSResult> {\n const { text, provider = \"kokoro\", voice = \"af_heart\" } = options;\n\n if (!text || typeof text !== \"string\" || text.trim().length === 0) {\n throw new Error(\"Text is required\");\n }\n\n switch (provider) {\n case \"kokoro\":\n return generateKokoroSpeech(text, voice);\n\n case \"deepgram\":\n return generateDeepgramSpeech(text, voice);\n\n default:\n throw new Error(`Unknown TTS provider: ${provider}`);\n }\n}\n"],"mappings":";;;AAiBA,eAAsB,EACpB,GACA,IAAgB,YAChB,IAA4C,CAAC,GACzB;CACpB,IAAM,IAAc,EAAc,SAAS,CAAoB,IAC1D,IACD;CAEJ,OAAO,EAAc,GAAM;EAAE,GAAG;EAAS,OAAO;CAAY,CAAC;AAC/D;;;ACbA,SAAS,IAA2B;CAClC,IAAM,IAAI;CACV,IAAI;EACF,IAAI,OAAO,EAAE,wBAAyB,YAAY;GAChD,IAAM,IAAM,EAAE,qBAAqB;GACnC,IAAI,GAAK,KAAK,IAAI,OAAO,EAAI,IAAI;EACnC;CACF,QAAQ,CAER;CACA,OAAO,EAAE,SAAS,MAAM,EAAE;AAC5B;AAKA,eAAsB,EACpB,GACA,IAAkB,SACE;CAEpB,IAAM,IAAY,EAAkB,SAAS,CAA0B,IAClE,IACD,SAEE,IAAK,EAAoB;CAE/B,IAAI,CAAC,GACH,MAAU,MAAM,qCAAqC;CASvD,OAAO;EACL,OAAO,MAPY,EAAG,IAAI,uBAAuB;GACjD,MAAM,EAAK,MAAM,GAAG,GAAI;GACxB,SAAS;GACT,UAAU;EACZ,CAAC;EAIC,aAAa;CACf;AACF;;;ACPA,eAAsB,EACpB,GACoB;CACpB,IAAM,EAAE,SAAM,cAAW,UAAU,WAAQ,eAAe;CAE1D,IAAI,CAAC,KAAQ,OAAO,KAAS,YAAY,EAAK,KAAK,CAAC,CAAC,WAAW,GAC9D,MAAU,MAAM,kBAAkB;CAGpC,QAAQ,GAAR;EACE,KAAK,UACH,OAAO,EAAqB,GAAM,CAAK;EAEzC,KAAK,YACH,OAAO,EAAuB,GAAM,CAAK;EAE3C,SACE,MAAU,MAAM,yBAAyB,GAAU;CACvD;AACF"}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
import { t as e } from "./semantic-split-DF26uo-Z.js";
|
|
2
|
+
//#region speech/types/types.ts
|
|
3
|
+
var t = /* @__PURE__ */ "af_heart.af_alloy.af_aoede.af_bella.af_jessica.af_kore.af_nicole.af_nova.af_river.af_sarah.af_sky.am_adam.am_echo.am_eric.am_fenrir.am_liam.am_michael.am_onyx.am_puck.am_santa.bf_alice.bf_emma.bf_isabella.bf_lily.bm_daniel.bm_fable.bm_george.bm_lewis".split(".");
|
|
4
|
+
function n(e) {
|
|
5
|
+
let [t, n] = e, r = e.slice(3) || e;
|
|
6
|
+
return {
|
|
7
|
+
id: e,
|
|
8
|
+
name: r.charAt(0).toUpperCase() + r.slice(1),
|
|
9
|
+
accent: t === "b" ? "British English" : "American English",
|
|
10
|
+
gender: n === "m" ? "Male" : "Female"
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
var r = [
|
|
14
|
+
"angus",
|
|
15
|
+
"asteria",
|
|
16
|
+
"arcas",
|
|
17
|
+
"orion",
|
|
18
|
+
"orpheus",
|
|
19
|
+
"athena",
|
|
20
|
+
"luna",
|
|
21
|
+
"zeus",
|
|
22
|
+
"perseus",
|
|
23
|
+
"helios",
|
|
24
|
+
"hera",
|
|
25
|
+
"stella"
|
|
26
|
+
], i = 44;
|
|
27
|
+
function a(e, t = 0) {
|
|
28
|
+
let n = Math.max(e.length - 1, 0) * Math.max(t, 0), r = e.reduce((e, t) => e + t.length, 0) + n, i = new Float32Array(r), a = 0;
|
|
29
|
+
return e.forEach((n, r) => {
|
|
30
|
+
i.set(n, a), a += n.length, r < e.length - 1 && (a += Math.max(t, 0));
|
|
31
|
+
}), i;
|
|
32
|
+
}
|
|
33
|
+
function o(e, t = 24e3, n = 1) {
|
|
34
|
+
let r = /* @__PURE__ */ new ArrayBuffer(i + e.length * 2), a = new DataView(r), o = (e, t) => {
|
|
35
|
+
for (let n = 0; n < t.length; n += 1) a.setUint8(e + n, t.charCodeAt(n));
|
|
36
|
+
}, s = t * n * 2;
|
|
37
|
+
o(0, "RIFF"), a.setUint32(4, 36 + e.length * 2, !0), o(8, "WAVE"), o(12, "fmt "), a.setUint32(16, 16, !0), a.setUint16(20, 1, !0), a.setUint16(22, n, !0), a.setUint32(24, t, !0), a.setUint32(28, s, !0), a.setUint16(32, n * 2, !0), a.setUint16(34, 16, !0), o(36, "data"), a.setUint32(40, e.length * 2, !0);
|
|
38
|
+
for (let t = 0; t < e.length; t += 1) {
|
|
39
|
+
let n = Math.max(-1, Math.min(1, e[t]));
|
|
40
|
+
a.setInt16(i + t * 2, n < 0 ? n * 32768 : n * 32767, !0);
|
|
41
|
+
}
|
|
42
|
+
return r;
|
|
43
|
+
}
|
|
44
|
+
function s(e, t) {
|
|
45
|
+
return t > 0 ? e.length / t : 0;
|
|
46
|
+
}
|
|
47
|
+
//#endregion
|
|
48
|
+
//#region speech/core/kokoro-node.ts
|
|
49
|
+
var c = "onnx-community/Kokoro-82M-v1.0-ONNX", l = "af_heart", u = "q8", d = "cpu", f = 400, p = 120, m = /* @__PURE__ */ new Map();
|
|
50
|
+
function h() {
|
|
51
|
+
let e = globalThis;
|
|
52
|
+
return !!(e.process?.versions?.node || e.Bun || e.Deno);
|
|
53
|
+
}
|
|
54
|
+
async function g(e = {}) {
|
|
55
|
+
let t = e.model ?? c, n = e.dtype ?? u, r = e.device ?? (h() ? d : "wasm"), i = `${t}|${n}|${r}`, a = m.get(i);
|
|
56
|
+
if (a) return a;
|
|
57
|
+
let o = (async () => {
|
|
58
|
+
let i;
|
|
59
|
+
try {
|
|
60
|
+
({KokoroTTS: i} = await import(
|
|
61
|
+
/* @vite-ignore */
|
|
62
|
+
"kokoro-js"
|
|
63
|
+
));
|
|
64
|
+
} catch (e) {
|
|
65
|
+
throw Error(`Local Kokoro speech needs the optional \`kokoro-js\` package. Install it with \`npm install kokoro-js\`. (import failed: ${e instanceof Error ? e.message : String(e)})`);
|
|
66
|
+
}
|
|
67
|
+
return i.from_pretrained(t, {
|
|
68
|
+
dtype: n,
|
|
69
|
+
device: r,
|
|
70
|
+
progress_callback: e.onModelProgress
|
|
71
|
+
});
|
|
72
|
+
})();
|
|
73
|
+
m.set(i, o);
|
|
74
|
+
try {
|
|
75
|
+
return await o;
|
|
76
|
+
} catch (e) {
|
|
77
|
+
throw m.delete(i), e;
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
function _() {
|
|
81
|
+
m.clear();
|
|
82
|
+
}
|
|
83
|
+
async function v(t, n = {}) {
|
|
84
|
+
let r = (t ?? "").trim();
|
|
85
|
+
if (!r) throw Error("Text is required");
|
|
86
|
+
let i = e(r, n.maxChunkLength ?? f).map((e) => e.trim()).filter((e) => e.length > 0), o = await g(n), s = n.voice ?? l, c = n.speed ?? 1, u = [], d = 24e3;
|
|
87
|
+
for (let e = 0; e < i.length; e += 1) {
|
|
88
|
+
n.onChunk?.({
|
|
89
|
+
index: e,
|
|
90
|
+
total: i.length,
|
|
91
|
+
text: i[e]
|
|
92
|
+
});
|
|
93
|
+
let t = await o.generate(i[e], {
|
|
94
|
+
voice: s,
|
|
95
|
+
speed: c
|
|
96
|
+
});
|
|
97
|
+
u.push(t.audio), d = t.sampling_rate ?? d;
|
|
98
|
+
}
|
|
99
|
+
let m = n.gapMs ?? p;
|
|
100
|
+
return {
|
|
101
|
+
samples: a(u, Math.round(m / 1e3 * d)),
|
|
102
|
+
sampleRate: d
|
|
103
|
+
};
|
|
104
|
+
}
|
|
105
|
+
async function y(e, t = {}) {
|
|
106
|
+
let { samples: n, sampleRate: r } = await v(e, t);
|
|
107
|
+
return {
|
|
108
|
+
audio: o(n, r),
|
|
109
|
+
contentType: "audio/wav"
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
//#endregion
|
|
113
|
+
export { a, r as c, y as i, t as l, _ as n, o, v as r, s, g as t, n as u };
|
|
114
|
+
|
|
115
|
+
//# sourceMappingURL=kokoro-node-CDhCeD6u.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"kokoro-node-CDhCeD6u.js","names":[],"sources":["../speech/types/types.ts","../speech/utils/wav.ts","../speech/core/kokoro-node.ts"],"sourcesContent":["/**\n * @fileoverview Type definitions for text-to-speech providers\n */\n\nexport type TTSProvider = \"kokoro\" | \"deepgram\";\n\nexport interface TTSOptions {\n text: string;\n provider?: TTSProvider;\n voice?: string;\n}\n\nexport interface TTSResult {\n audio: ArrayBuffer;\n contentType: string;\n}\n\n// Kokoro voices from the model. The `a`/`b` prefix is the accent (American /\n// British English) and the `f`/`m` that follows it is the voice's gender, which\n// is how `describeKokoroVoice` derives a listing without a second catalog.\nexport const KOKORO_VOICES = [\n \"af_heart\", \"af_alloy\", \"af_aoede\", \"af_bella\", \"af_jessica\",\n \"af_kore\", \"af_nicole\", \"af_nova\", \"af_river\", \"af_sarah\", \"af_sky\",\n \"am_adam\", \"am_echo\", \"am_eric\", \"am_fenrir\", \"am_liam\",\n \"am_michael\", \"am_onyx\", \"am_puck\", \"am_santa\",\n \"bf_alice\", \"bf_emma\", \"bf_isabella\", \"bf_lily\",\n \"bm_daniel\", \"bm_fable\", \"bm_george\", \"bm_lewis\",\n] as const;\n\nexport type KokoroVoice = (typeof KOKORO_VOICES)[number];\n\nexport interface KokoroVoiceDescription {\n id: string;\n /** Display name, e.g. `Heart` for `af_heart`. */\n name: string;\n /** `American English` or `British English`. */\n accent: string;\n gender: \"Female\" | \"Male\";\n}\n\n/**\n * Describes a voice from its id, so listings (`--list-voices`, a voice picker)\n * do not need a hand-maintained table that can drift from `KOKORO_VOICES`.\n */\nexport function describeKokoroVoice(id: string): KokoroVoiceDescription {\n const [accentCode, genderCode] = id;\n const suffix = id.slice(3) || id;\n\n return {\n id,\n name: suffix.charAt(0).toUpperCase() + suffix.slice(1),\n accent: accentCode === \"b\" ? \"British English\" : \"American English\",\n gender: genderCode === \"m\" ? \"Male\" : \"Female\",\n };\n}\n\n// Deepgram Aura speakers\nexport const DEEPGRAM_SPEAKERS = [\n \"angus\", \"asteria\", \"arcas\", \"orion\", \"orpheus\", \"athena\",\n \"luna\", \"zeus\", \"perseus\", \"helios\", \"hera\", \"stella\",\n] as const;\n\nexport type DeepgramSpeaker = (typeof DEEPGRAM_SPEAKERS)[number];\n","/**\n * @fileoverview Minimal WAV (RIFF) encoder for raw Float32 PCM.\n *\n * The browser path gets its WAV from `convertAudioBufferToWav`, which needs a Web\n * Audio `AudioBuffer`. Node has no such thing: the Kokoro model hands back a bare\n * `Float32Array`, and long documents are synthesized one chunk at a time and\n * joined before encoding. Both of those are pure buffer work, so they live here\n * with no platform dependency.\n */\n\n/** Number of bytes in the fixed RIFF/fmt/data header this encoder writes. */\nconst HEADER_BYTES = 44;\n\n/**\n * Joins per-chunk sample buffers into one contiguous track.\n *\n * @param chunks Sample buffers in playback order.\n * @param silenceSamples Samples of silence to insert between chunks, so the\n * seams between separately synthesized chunks do not sound clipped.\n */\nexport function concatSamples(\n chunks: Float32Array[],\n silenceSamples = 0\n): Float32Array {\n const gaps = Math.max(chunks.length - 1, 0) * Math.max(silenceSamples, 0);\n const total = chunks.reduce((sum, chunk) => sum + chunk.length, 0) + gaps;\n const out = new Float32Array(total);\n\n let offset = 0;\n chunks.forEach((chunk, i) => {\n out.set(chunk, offset);\n offset += chunk.length;\n if (i < chunks.length - 1) offset += Math.max(silenceSamples, 0);\n });\n\n return out;\n}\n\n/**\n * Encodes Float32 samples (nominally -1..1) as a 16-bit PCM WAV file.\n *\n * @param samples Interleaved samples when `channels` is greater than 1.\n * @param sampleRate Sample rate of the audio, e.g. 24000 for Kokoro.\n * @param channels Channel count. Kokoro is mono.\n */\nexport function encodeWav(\n samples: Float32Array,\n sampleRate = 24000,\n channels = 1\n): ArrayBuffer {\n const buffer = new ArrayBuffer(HEADER_BYTES + samples.length * 2);\n const view = new DataView(buffer);\n\n const writeString = (offset: number, value: string): void => {\n for (let i = 0; i < value.length; i += 1) {\n view.setUint8(offset + i, value.charCodeAt(i));\n }\n };\n\n const byteRate = sampleRate * channels * 2;\n\n writeString(0, \"RIFF\");\n view.setUint32(4, 36 + samples.length * 2, true);\n writeString(8, \"WAVE\");\n writeString(12, \"fmt \");\n view.setUint32(16, 16, true); // PCM header size\n view.setUint16(20, 1, true); // format: PCM\n view.setUint16(22, channels, true);\n view.setUint32(24, sampleRate, true);\n view.setUint32(28, byteRate, true);\n view.setUint16(32, channels * 2, true); // block align\n view.setUint16(34, 16, true); // bits per sample\n writeString(36, \"data\");\n view.setUint32(40, samples.length * 2, true);\n\n for (let i = 0; i < samples.length; i += 1) {\n // Clamp before scaling: the model occasionally overshoots 1.0 and wrapping\n // a 16-bit sample turns that into a loud click.\n const sample = Math.max(-1, Math.min(1, samples[i]));\n view.setInt16(HEADER_BYTES + i * 2, sample < 0 ? sample * 0x8000 : sample * 0x7fff, true);\n }\n\n return buffer;\n}\n\n/** Duration of a Float32 track in seconds, for progress and log lines. */\nexport function wavDurationSeconds(samples: Float32Array, sampleRate: number): number {\n return sampleRate > 0 ? samples.length / sampleRate : 0;\n}\n","/**\n * @fileoverview Kokoro TTS running locally on Node (or Bun/Deno) via `kokoro-js`.\n *\n * This is the engine behind `generateSpeech({ provider: \"kokoro\" })` on a server\n * and behind the `use-voice-control` CLI. The model runs on the CPU through\n * onnxruntime, so nothing is sent to a third-party service; the first call\n * downloads the weights (~90 MB at the default `q8`) into the Hugging Face cache\n * and every later call reuses the instance held in this module.\n *\n * Kokoro's context is a few hundred phonemes, so anything longer than a\n * paragraph has to be synthesized in pieces. `synthesizeSamples` chunks on\n * sentence and paragraph boundaries with `splitTextSmart`, runs the chunks in\n * order, and joins the audio with a short silence at each seam.\n */\nimport type { TTSResult } from \"../types/types\";\nimport { splitTextSmart } from \"../utils/semantic-split.js\";\nimport { concatSamples, encodeWav } from \"../utils/wav\";\n\nexport type KokoroDtype = \"fp32\" | \"fp16\" | \"q8\" | \"q4\" | \"q4f16\";\nexport type KokoroDevice = \"wasm\" | \"webgpu\" | \"cpu\";\n\nexport interface KokoroNodeOptions {\n /** Voice id, e.g. `af_heart`. Default `af_heart`. */\n voice?: string;\n /** Speaking rate; 1 is the model's natural pace. Default 1. */\n speed?: number;\n /** Hugging Face model id. Default `onnx-community/Kokoro-82M-v1.0-ONNX`. */\n model?: string;\n /** Weight quantization. `q8` (default) is the best size/quality trade on CPU. */\n dtype?: KokoroDtype;\n /** Execution device. Default `cpu`. */\n device?: KokoroDevice;\n /**\n * Target chunk size in characters. Kokoro's context is limited, so long text\n * is split before synthesis; 400 leaves headroom for phoneme expansion.\n */\n maxChunkLength?: number;\n /** Silence inserted between chunks, in milliseconds. Default 120. */\n gapMs?: number;\n /** Model download/loading progress, forwarded from transformers.js. */\n onModelProgress?: (progress: unknown) => void;\n /** Called before each chunk is synthesized, for CLI progress output. */\n onChunk?: (info: { index: number; total: number; text: string }) => void;\n}\n\nexport interface KokoroAudio {\n samples: Float32Array;\n sampleRate: number;\n}\n\nconst DEFAULT_MODEL = \"onnx-community/Kokoro-82M-v1.0-ONNX\";\nconst DEFAULT_VOICE = \"af_heart\";\nconst DEFAULT_DTYPE: KokoroDtype = \"q8\";\nconst DEFAULT_DEVICE: KokoroDevice = \"cpu\";\nconst DEFAULT_MAX_CHUNK = 400;\nconst DEFAULT_GAP_MS = 120;\n\n/** One loaded model per model/dtype/device combination, keyed by those three. */\nconst instances = new Map<string, Promise<any>>();\n\n/** True on Node, Bun and Deno — anywhere `kokoro-js` can load onnxruntime-node. */\nfunction isServerRuntime(): boolean {\n const g = globalThis as any;\n return Boolean(g.process?.versions?.node || g.Bun || g.Deno);\n}\n\n/**\n * Loads (and caches) a `kokoro-js` instance.\n *\n * `kokoro-js` is an optional dependency: a browser-only consumer should not have\n * to download onnxruntime-node, so a missing install is reported as an\n * actionable error rather than a module-resolution stack trace.\n */\nexport async function loadKokoroTTS(options: KokoroNodeOptions = {}): Promise<any> {\n const model = options.model ?? DEFAULT_MODEL;\n const dtype = options.dtype ?? DEFAULT_DTYPE;\n const device = options.device ?? (isServerRuntime() ? DEFAULT_DEVICE : \"wasm\");\n const key = `${model}|${dtype}|${device}`;\n\n const cached = instances.get(key);\n if (cached) return cached;\n\n const loading = (async () => {\n let KokoroTTS: any;\n try {\n // Kept in a variable so bundlers treat this as an optional runtime\n // dependency rather than something to resolve at build time.\n const specifier = \"kokoro-js\";\n ({ KokoroTTS } = await import(/* @vite-ignore */ specifier));\n } catch (error) {\n throw new Error(\n \"Local Kokoro speech needs the optional `kokoro-js` package. \" +\n \"Install it with `npm install kokoro-js`. \" +\n `(import failed: ${error instanceof Error ? error.message : String(error)})`\n );\n }\n\n return KokoroTTS.from_pretrained(model, {\n dtype,\n device,\n progress_callback: options.onModelProgress,\n });\n })();\n\n instances.set(key, loading);\n\n try {\n return await loading;\n } catch (error) {\n // A failed download must not poison every later attempt.\n instances.delete(key);\n throw error;\n }\n}\n\n/** Drops cached model instances. Mainly useful in tests and long-lived workers. */\nexport function resetKokoroCache(): void {\n instances.clear();\n}\n\n/**\n * Synthesizes `text` to raw samples, chunking anything longer than the model's\n * comfortable context.\n */\nexport async function synthesizeSamples(\n text: string,\n options: KokoroNodeOptions = {}\n): Promise<KokoroAudio> {\n const trimmed = (text ?? \"\").trim();\n if (!trimmed) throw new Error(\"Text is required\");\n\n const chunks = splitTextSmart(trimmed, options.maxChunkLength ?? DEFAULT_MAX_CHUNK)\n .map((chunk: string) => chunk.trim())\n .filter((chunk: string) => chunk.length > 0);\n\n const tts = await loadKokoroTTS(options);\n const voice = options.voice ?? DEFAULT_VOICE;\n const speed = options.speed ?? 1;\n\n const rendered: Float32Array[] = [];\n let sampleRate = 24000;\n\n for (let index = 0; index < chunks.length; index += 1) {\n options.onChunk?.({ index, total: chunks.length, text: chunks[index] });\n const audio = await tts.generate(chunks[index], { voice, speed });\n rendered.push(audio.audio);\n sampleRate = audio.sampling_rate ?? sampleRate;\n }\n\n const gapMs = options.gapMs ?? DEFAULT_GAP_MS;\n return {\n samples: concatSamples(rendered, Math.round((gapMs / 1000) * sampleRate)),\n sampleRate,\n };\n}\n\n/** Synthesizes `text` and encodes it as a 16-bit PCM WAV file. */\nexport async function synthesizeWav(\n text: string,\n options: KokoroNodeOptions = {}\n): Promise<TTSResult> {\n const { samples, sampleRate } = await synthesizeSamples(text, options);\n return { audio: encodeWav(samples, sampleRate), contentType: \"audio/wav\" };\n}\n"],"mappings":";;AAoBA,IAAa,IAAgB,sRAO7B;AAiBA,SAAgB,EAAoB,GAAoC;CACtE,IAAM,CAAC,GAAY,KAAc,GAC3B,IAAS,EAAG,MAAM,CAAC,KAAK;CAE9B,OAAO;EACL;EACA,MAAM,EAAO,OAAO,CAAC,CAAC,CAAC,YAAY,IAAI,EAAO,MAAM,CAAC;EACrD,QAAQ,MAAe,MAAM,oBAAoB;EACjD,QAAQ,MAAe,MAAM,SAAS;CACxC;AACF;AAGA,IAAa,IAAoB;CAC/B;CAAS;CAAW;CAAS;CAAS;CAAW;CACjD;CAAQ;CAAQ;CAAW;CAAU;CAAQ;AAC/C,GCjDM,IAAe;AASrB,SAAgB,EACd,GACA,IAAiB,GACH;CACd,IAAM,IAAO,KAAK,IAAI,EAAO,SAAS,GAAG,CAAC,IAAI,KAAK,IAAI,GAAgB,CAAC,GAClE,IAAQ,EAAO,QAAQ,GAAK,MAAU,IAAM,EAAM,QAAQ,CAAC,IAAI,GAC/D,IAAM,IAAI,aAAa,CAAK,GAE9B,IAAS;CAOb,OANA,EAAO,SAAS,GAAO,MAAM;EAG3B,AAFA,EAAI,IAAI,GAAO,CAAM,GACrB,KAAU,EAAM,QACZ,IAAI,EAAO,SAAS,MAAG,KAAU,KAAK,IAAI,GAAgB,CAAC;CACjE,CAAC,GAEM;AACT;AASA,SAAgB,EACd,GACA,IAAa,MACb,IAAW,GACE;CACb,IAAM,oBAAS,IAAI,YAAY,IAAe,EAAQ,SAAS,CAAC,GAC1D,IAAO,IAAI,SAAS,CAAM,GAE1B,KAAe,GAAgB,MAAwB;EAC3D,KAAK,IAAI,IAAI,GAAG,IAAI,EAAM,QAAQ,KAAK,GACrC,EAAK,SAAS,IAAS,GAAG,EAAM,WAAW,CAAC,CAAC;CAEjD,GAEM,IAAW,IAAa,IAAW;CAczC,AAZA,EAAY,GAAG,MAAM,GACrB,EAAK,UAAU,GAAG,KAAK,EAAQ,SAAS,GAAG,EAAI,GAC/C,EAAY,GAAG,MAAM,GACrB,EAAY,IAAI,MAAM,GACtB,EAAK,UAAU,IAAI,IAAI,EAAI,GAC3B,EAAK,UAAU,IAAI,GAAG,EAAI,GAC1B,EAAK,UAAU,IAAI,GAAU,EAAI,GACjC,EAAK,UAAU,IAAI,GAAY,EAAI,GACnC,EAAK,UAAU,IAAI,GAAU,EAAI,GACjC,EAAK,UAAU,IAAI,IAAW,GAAG,EAAI,GACrC,EAAK,UAAU,IAAI,IAAI,EAAI,GAC3B,EAAY,IAAI,MAAM,GACtB,EAAK,UAAU,IAAI,EAAQ,SAAS,GAAG,EAAI;CAE3C,KAAK,IAAI,IAAI,GAAG,IAAI,EAAQ,QAAQ,KAAK,GAAG;EAG1C,IAAM,IAAS,KAAK,IAAI,IAAI,KAAK,IAAI,GAAG,EAAQ,EAAE,CAAC;EACnD,EAAK,SAAS,IAAe,IAAI,GAAG,IAAS,IAAI,IAAS,QAAS,IAAS,OAAQ,EAAI;CAC1F;CAEA,OAAO;AACT;AAGA,SAAgB,EAAmB,GAAuB,GAA4B;CACpF,OAAO,IAAa,IAAI,EAAQ,SAAS,IAAa;AACxD;;;ACtCA,IAAM,IAAgB,uCAChB,IAAgB,YAChB,IAA6B,MAC7B,IAA+B,OAC/B,IAAoB,KACpB,IAAiB,KAGjB,oBAAY,IAAI,IAA0B;AAGhD,SAAS,IAA2B;CAClC,IAAM,IAAI;CACV,OAAO,GAAQ,EAAE,SAAS,UAAU,QAAQ,EAAE,OAAO,EAAE;AACzD;AASA,eAAsB,EAAc,IAA6B,CAAC,GAAiB;CACjF,IAAM,IAAQ,EAAQ,SAAS,GACzB,IAAQ,EAAQ,SAAS,GACzB,IAAS,EAAQ,WAAW,EAAgB,IAAI,IAAiB,SACjE,IAAM,GAAG,EAAM,GAAG,EAAM,GAAG,KAE3B,IAAS,EAAU,IAAI,CAAG;CAChC,IAAI,GAAQ,OAAO;CAEnB,IAAM,KAAW,YAAY;EAC3B,IAAI;EACJ,IAAI;GAIF,CAAC,iBAAgB,MAAM;;IAA0B;;EACnD,SAAS,GAAO;GACd,MAAU,MACR,4HAEqB,aAAiB,QAAQ,EAAM,UAAU,OAAO,CAAK,EAAE,EAC9E;EACF;EAEA,OAAO,EAAU,gBAAgB,GAAO;GACtC;GACA;GACA,mBAAmB,EAAQ;EAC7B,CAAC;CACH,EAAA,CAAG;CAEH,EAAU,IAAI,GAAK,CAAO;CAE1B,IAAI;EACF,OAAO,MAAM;CACf,SAAS,GAAO;EAGd,MADA,EAAU,OAAO,CAAG,GACd;CACR;AACF;AAGA,SAAgB,IAAyB;CACvC,EAAU,MAAM;AAClB;AAMA,eAAsB,EACpB,GACA,IAA6B,CAAC,GACR;CACtB,IAAM,KAAW,KAAQ,GAAA,CAAI,KAAK;CAClC,IAAI,CAAC,GAAS,MAAU,MAAM,kBAAkB;CAEhD,IAAM,IAAS,EAAe,GAAS,EAAQ,kBAAkB,CAAiB,CAAC,CAChF,KAAK,MAAkB,EAAM,KAAK,CAAC,CAAC,CACpC,QAAQ,MAAkB,EAAM,SAAS,CAAC,GAEvC,IAAM,MAAM,EAAc,CAAO,GACjC,IAAQ,EAAQ,SAAS,GACzB,IAAQ,EAAQ,SAAS,GAEzB,IAA2B,CAAC,GAC9B,IAAa;CAEjB,KAAK,IAAI,IAAQ,GAAG,IAAQ,EAAO,QAAQ,KAAS,GAAG;EACrD,EAAQ,UAAU;GAAE;GAAO,OAAO,EAAO;GAAQ,MAAM,EAAO;EAAO,CAAC;EACtE,IAAM,IAAQ,MAAM,EAAI,SAAS,EAAO,IAAQ;GAAE;GAAO;EAAM,CAAC;EAEhE,AADA,EAAS,KAAK,EAAM,KAAK,GACzB,IAAa,EAAM,iBAAiB;CACtC;CAEA,IAAM,IAAQ,EAAQ,SAAS;CAC/B,OAAO;EACL,SAAS,EAAc,GAAU,KAAK,MAAO,IAAQ,MAAQ,CAAU,CAAC;EACxE;CACF;AACF;AAGA,eAAsB,EACpB,GACA,IAA6B,CAAC,GACV;CACpB,IAAM,EAAE,YAAS,kBAAe,MAAM,EAAkB,GAAM,CAAO;CACrE,OAAO;EAAE,OAAO,EAAU,GAAS,CAAU;EAAG,aAAa;CAAY;AAC3E"}
|