use-voice-control 0.1.86 → 0.1.87

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +206 -21
  2. package/bin/use-voice-control.mjs +20 -0
  3. package/dist/cli.js +374 -0
  4. package/dist/cli.js.map +1 -0
  5. package/dist/client/index.d.ts +1 -0
  6. package/dist/client/read-aloud.d.ts +14 -0
  7. package/dist/client.js +132 -157
  8. package/dist/client.js.map +1 -1
  9. package/dist/core/kokoro-node.d.ts +51 -0
  10. package/dist/core/kokoro.d.ts +6 -2
  11. package/dist/index.d.ts +2 -0
  12. package/dist/index.js +33 -76
  13. package/dist/index.js.map +1 -1
  14. package/dist/kokoro-node-DJ_Rxp_N.js +144 -0
  15. package/dist/kokoro-node-DJ_Rxp_N.js.map +1 -0
  16. package/dist/markdown.js +213 -0
  17. package/dist/markdown.js.map +1 -0
  18. package/dist/node/cli.d.ts +52 -0
  19. package/dist/node/document.d.ts +45 -0
  20. package/dist/node/index.d.ts +18 -0
  21. package/dist/node/render.d.ts +30 -0
  22. package/dist/node.js +27 -0
  23. package/dist/node.js.map +1 -0
  24. package/dist/react.js +13 -11
  25. package/dist/react.js.map +1 -1
  26. package/dist/semantic-split-CXhk-k1F.js +44 -0
  27. package/dist/semantic-split-CXhk-k1F.js.map +1 -0
  28. package/dist/types/types.d.ts +14 -1
  29. package/dist/utils/markdown-to-speech.d.ts +76 -0
  30. package/dist/utils/wav.d.ts +27 -0
  31. package/package.json +17 -4
  32. package/speech/client/index.ts +10 -0
  33. package/speech/client/read-aloud.ts +27 -1
  34. package/speech/core/kokoro-node.ts +164 -0
  35. package/speech/core/kokoro.ts +13 -66
  36. package/speech/index.ts +14 -0
  37. package/speech/node/cli.ts +509 -0
  38. package/speech/node/document.ts +126 -0
  39. package/speech/node/index.ts +64 -0
  40. package/speech/node/render.ts +79 -0
  41. package/speech/react/useReadAloud.ts +2 -0
  42. package/speech/types/types.ts +34 -5
  43. package/speech/utils/markdown-to-speech.ts +426 -0
  44. package/speech/utils/wav.ts +89 -0
@@ -0,0 +1,44 @@
1
+ function h(f, i = 500) {
2
+ const n = f.split(/\n\s*\n/), s = [];
3
+ for (let o of n) {
4
+ if (o.length <= i) {
5
+ s.push(o.trim());
6
+ continue;
7
+ }
8
+ const r = new RegExp(`(?<=[.?!])(?=\\s+["“”'a-z])`, "gi"), l = o.split(r);
9
+ let t = "";
10
+ for (let e of l) {
11
+ if (e = e.trim(), e.length > i) {
12
+ const p = u(e, i);
13
+ for (let c of p)
14
+ (t + " " + c).length > i ? (t && s.push(t.trim()), t = c) : t += (t ? " " : "") + c;
15
+ continue;
16
+ }
17
+ (t + " " + e).length > i ? (t && s.push(t.trim()), t = e) : t += (t ? " " : "") + e;
18
+ }
19
+ t && s.push(t.trim());
20
+ }
21
+ return s;
22
+ }
23
+ function u(f, i) {
24
+ const n = [];
25
+ let s = "";
26
+ const o = f.split(/,\s*/);
27
+ for (let r of o)
28
+ if ((s + ", " + r).length > i)
29
+ if (s && n.push(s.trim()), r.length > i) {
30
+ const l = r.split(/\s+/);
31
+ let t = "";
32
+ for (let e of l)
33
+ (t + " " + e).length > i ? (t && n.push(t.trim()), t = e) : t += (t ? " " : "") + e;
34
+ t && n.push(t.trim()), s = "";
35
+ } else
36
+ s = r;
37
+ else
38
+ s += (s ? ", " : "") + r;
39
+ return s && n.push(s.trim()), n;
40
+ }
41
+ export {
42
+ h as s
43
+ };
44
+ //# sourceMappingURL=semantic-split-CXhk-k1F.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"semantic-split-CXhk-k1F.js","sources":["../speech/utils/semantic-split.js"],"sourcesContent":["/**\n * @fileoverview Splits long text into TTS-sized chunks along paragraph, sentence, comma, and word boundaries so speech synthesis can process it incrementally.\n *\n * Exports `splitTextSmart` (primary chunker, falling back to `splitLongSentence` for\n * oversized sentences) and `splitLongSentence`. Also retains an older, unused\n * `splitTextSmartOld` variant that lacks the long-sentence fallback.\n */\nexport function splitTextSmart(text, maxChunkLength = 500) {\n const paragraphChunks = text.split(/\\n\\s*\\n/);\n const finalChunks = [];\n\n for (let para of paragraphChunks) {\n if (para.length <= maxChunkLength) {\n finalChunks.push(para.trim());\n continue;\n }\n\n const sentenceRegex = /(?<=[.?!])(?=\\s+[\"“”'a-z])/gi;\n const sentences = para.split(sentenceRegex);\n\n let chunk = '';\n for (let sentence of sentences) {\n sentence = sentence.trim();\n\n if (sentence.length > maxChunkLength) {\n // Sentence too long — fallback split\n const subChunks = splitLongSentence(sentence, maxChunkLength);\n for (let sub of subChunks) {\n if ((chunk + ' ' + sub).length > maxChunkLength) {\n if (chunk) finalChunks.push(chunk.trim());\n chunk = sub;\n } else {\n chunk += (chunk ? ' ' : '') + sub;\n }\n }\n continue;\n }\n\n if ((chunk + ' ' + sentence).length > maxChunkLength) {\n if (chunk) finalChunks.push(chunk.trim());\n chunk = sentence;\n } else {\n chunk += (chunk ? ' ' : '') + sentence;\n }\n }\n if (chunk) finalChunks.push(chunk.trim());\n }\n\n return finalChunks;\n}\n\nexport function splitLongSentence(sentence, maxLen) {\n const chunks = [];\n let current = '';\n\n const commaParts = sentence.split(/,\\s*/);\n for (let part of commaParts) {\n if ((current + ', ' + part).length > maxLen) {\n if (current) chunks.push(current.trim());\n if (part.length > maxLen) {\n const words = part.split(/\\s+/);\n let wordChunk = '';\n for (let word of words) {\n if ((wordChunk + ' ' + word).length > maxLen) {\n if (wordChunk) chunks.push(wordChunk.trim());\n wordChunk = word;\n } else {\n wordChunk += (wordChunk ? ' ' : '') + word;\n }\n }\n if (wordChunk) chunks.push(wordChunk.trim());\n current = '';\n } else {\n current = part;\n }\n } else {\n current += (current ? ', ' : '') + part;\n }\n }\n if (current) chunks.push(current.trim());\n\n return chunks;\n}\n\n\nfunction splitTextSmartOld(text, maxChunkLength = 500) {\n const paragraphChunks = text.split(/\\n\\s*\\n/); // Step 1: split on double returns\n const finalChunks = [];\n\n for (let para of paragraphChunks) {\n if (para.length <= maxChunkLength) {\n finalChunks.push(para.trim());\n continue;\n }\n\n // Step 2: Further split on sentence boundaries if too long\n const sentenceRegex = /(?<=[.?!])(?=\\s+[\"“”'a-z])/gi;\n const sentences = para.split(sentenceRegex);\n\n let chunk = '';\n for (let sentence of sentences) {\n sentence = sentence.trim();\n if ((chunk + ' ' + sentence).length > maxChunkLength) {\n if (chunk) finalChunks.push(chunk.trim());\n chunk = sentence;\n } else {\n chunk += (chunk ? ' ' : '') + sentence;\n }\n }\n if (chunk) finalChunks.push(chunk.trim());\n }\n\n return finalChunks;\n}\n"],"names":["splitTextSmart","text","maxChunkLength","paragraphChunks","finalChunks","para","sentenceRegex","sentences","chunk","sentence","subChunks","splitLongSentence","sub","maxLen","chunks","current","commaParts","part","words","wordChunk","word"],"mappings":"AAOO,SAASA,EAAeC,GAAMC,IAAiB,KAAK;AACzD,QAAMC,IAAkBF,EAAK,MAAM,SAAS,GACtCG,IAAc,CAAA;AAEpB,WAASC,KAAQF,GAAiB;AAChC,QAAIE,EAAK,UAAUH,GAAgB;AACjC,MAAAE,EAAY,KAAKC,EAAK,MAAM;AAC5B;AAAA,IACF;AAEA,UAAMC,IAAgB,WAAA,+BAAA,IAA8B,GAC9CC,IAAYF,EAAK,MAAMC,CAAa;AAE1C,QAAIE,IAAQ;AACZ,aAASC,KAAYF,GAAW;AAG9B,UAFAE,IAAWA,EAAS,KAAI,GAEpBA,EAAS,SAASP,GAAgB;AAEpC,cAAMQ,IAAYC,EAAkBF,GAAUP,CAAc;AAC5D,iBAASU,KAAOF;AACd,WAAKF,IAAQ,MAAMI,GAAK,SAASV,KAC3BM,KAAOJ,EAAY,KAAKI,EAAM,KAAI,CAAE,GACxCA,IAAQI,KAERJ,MAAUA,IAAQ,MAAM,MAAMI;AAGlC;AAAA,MACF;AAEA,OAAKJ,IAAQ,MAAMC,GAAU,SAASP,KAChCM,KAAOJ,EAAY,KAAKI,EAAM,KAAI,CAAE,GACxCA,IAAQC,KAERD,MAAUA,IAAQ,MAAM,MAAMC;AAAA,IAElC;AACA,IAAID,KAAOJ,EAAY,KAAKI,EAAM,KAAI,CAAE;AAAA,EAC1C;AAEA,SAAOJ;AACT;AAEO,SAASO,EAAkBF,GAAUI,GAAQ;AAClD,QAAMC,IAAS,CAAA;AACf,MAAIC,IAAU;AAEd,QAAMC,IAAaP,EAAS,MAAM,MAAM;AACxC,WAASQ,KAAQD;AACf,SAAKD,IAAU,OAAOE,GAAM,SAASJ;AAEnC,UADIE,KAASD,EAAO,KAAKC,EAAQ,KAAI,CAAE,GACnCE,EAAK,SAASJ,GAAQ;AACxB,cAAMK,IAAQD,EAAK,MAAM,KAAK;AAC9B,YAAIE,IAAY;AAChB,iBAASC,KAAQF;AACf,WAAKC,IAAY,MAAMC,GAAM,SAASP,KAChCM,KAAWL,EAAO,KAAKK,EAAU,KAAI,CAAE,GAC3CA,IAAYC,KAEZD,MAAcA,IAAY,MAAM,MAAMC;AAG1C,QAAID,KAAWL,EAAO,KAAKK,EAAU,KAAI,CAAE,GAC3CJ,IAAU;AAAA,MACZ;AACE,QAAAA,IAAUE;AAAA;AAGZ,MAAAF,MAAYA,IAAU,OAAO,MAAME;AAGvC,SAAIF,KAASD,EAAO,KAAKC,EAAQ,KAAI,CAAE,GAEhCD;AACT;"}
@@ -11,7 +11,20 @@ export interface TTSResult {
11
11
  audio: ArrayBuffer;
12
12
  contentType: string;
13
13
  }
14
- export declare const KOKORO_VOICES: readonly ["af_heart", "af_alloy", "af_aoede", "af_bella", "af_jessica", "af_nicole", "af_river", "af_sarah", "af_sky", "am_adam", "am_echo", "am_fable", "am_fenrir", "am_liam", "am_michael", "am_onyx"];
14
+ export declare const KOKORO_VOICES: readonly ["af_heart", "af_alloy", "af_aoede", "af_bella", "af_jessica", "af_kore", "af_nicole", "af_nova", "af_river", "af_sarah", "af_sky", "am_adam", "am_echo", "am_eric", "am_fenrir", "am_liam", "am_michael", "am_onyx", "am_puck", "am_santa", "bf_alice", "bf_emma", "bf_isabella", "bf_lily", "bm_daniel", "bm_fable", "bm_george", "bm_lewis"];
15
15
  export type KokoroVoice = (typeof KOKORO_VOICES)[number];
16
+ export interface KokoroVoiceDescription {
17
+ id: string;
18
+ /** Display name, e.g. `Heart` for `af_heart`. */
19
+ name: string;
20
+ /** `American English` or `British English`. */
21
+ accent: string;
22
+ gender: "Female" | "Male";
23
+ }
24
+ /**
25
+ * Describes a voice from its id, so listings (`--list-voices`, a voice picker)
26
+ * do not need a hand-maintained table that can drift from `KOKORO_VOICES`.
27
+ */
28
+ export declare function describeKokoroVoice(id: string): KokoroVoiceDescription;
16
29
  export declare const DEEPGRAM_SPEAKERS: readonly ["angus", "asteria", "arcas", "orion", "orpheus", "athena", "luna", "zeus", "perseus", "helios", "hera", "stella"];
17
30
  export type DeepgramSpeaker = (typeof DEEPGRAM_SPEAKERS)[number];
@@ -0,0 +1,76 @@
1
+ /**
2
+ * @fileoverview Turns Markdown into text a speech synthesizer can read.
3
+ *
4
+ * Markdown handed straight to a TTS engine is read literally: "hash hash Getting
5
+ * started", "star star important star star". This module parses the document
6
+ * instead and emits only the words, keeping the *structure* the marks encoded —
7
+ * a heading becomes its own spoken segment ending in a full stop so the
8
+ * synthesizer pauses, a list item becomes one sentence, a fenced code block is
9
+ * announced rather than spelled out.
10
+ *
11
+ * It is deliberately dependency-free and runs in the browser as well as Node, so
12
+ * `ReadAloudController` and the CLI can share it.
13
+ */
14
+ /** Kind of block the converter recognised. */
15
+ export type SpeechSegmentType = "heading" | "paragraph" | "list-item" | "quote" | "code" | "table-row";
16
+ export interface SpeechSegment {
17
+ type: SpeechSegmentType;
18
+ /** Spoken text of the block, with every Markdown mark already removed. */
19
+ text: string;
20
+ /** Heading level 1-6. Only set on `heading` segments. */
21
+ level?: number;
22
+ }
23
+ export interface MarkdownToSpeechOptions {
24
+ /**
25
+ * How headings are spoken. `text` (default) reads the heading words on their
26
+ * own so they land between pauses; `announce` prefixes them with "Heading:";
27
+ * `skip` drops them entirely.
28
+ */
29
+ headings?: "text" | "announce" | "skip";
30
+ /**
31
+ * What to do with fenced code blocks. `announce` (default) replaces the block
32
+ * with a short spoken note, `read` reads the code verbatim, `skip` drops it.
33
+ */
34
+ codeBlocks?: "announce" | "read" | "skip";
35
+ /** `text` (default) speaks the link text and drops the URL; `text-and-url` reads both. */
36
+ links?: "text" | "text-and-url";
37
+ /** `alt` (default) speaks the image's alt text; `skip` drops images. */
38
+ images?: "alt" | "skip";
39
+ /** Read the YAML front matter block. Off by default. */
40
+ frontMatter?: boolean;
41
+ /** `rows` (default) reads table rows as comma-separated cells; `skip` drops tables. */
42
+ tables?: "rows" | "skip";
43
+ /**
44
+ * Append a full stop to blocks that do not end in punctuation, so the
45
+ * synthesizer pauses between them instead of running them together.
46
+ * Default true.
47
+ */
48
+ addTerminalPunctuation?: boolean;
49
+ }
50
+ /**
51
+ * Strips the inline marks — emphasis, code spans, links, images, raw HTML — from
52
+ * a single line, leaving the words behind.
53
+ */
54
+ export declare function stripInlineMarkdown(input: string, options?: MarkdownToSpeechOptions): string;
55
+ /**
56
+ * Parses Markdown into the blocks that should be spoken, in reading order.
57
+ *
58
+ * Use this when the caller wants the structure — to highlight the current
59
+ * heading, say, or to skip to a section. Callers that only need something to
60
+ * feed a synthesizer want {@link markdownToSpeech}.
61
+ */
62
+ export declare function markdownToSpeechSegments(markdown: string, options?: MarkdownToSpeechOptions): SpeechSegment[];
63
+ /**
64
+ * Converts Markdown to plain text ready for a speech synthesizer.
65
+ *
66
+ * Blocks are separated by blank lines so downstream chunkers (`splitTextSmart`)
67
+ * break on them; runs of list items and table rows stay in one block so a list
68
+ * is not chopped into one request per bullet.
69
+ */
70
+ export declare function markdownToSpeech(markdown: string, options?: MarkdownToSpeechOptions): string;
71
+ /**
72
+ * Guesses whether a blob of text is Markdown, for callers with no filename to go
73
+ * on (piped stdin, a paste). Errs towards `false`: prose read as Markdown is
74
+ * mostly unchanged anyway, so only reasonably clear signals count.
75
+ */
76
+ export declare function looksLikeMarkdown(text: string): boolean;
@@ -0,0 +1,27 @@
1
+ /**
2
+ * @fileoverview Minimal WAV (RIFF) encoder for raw Float32 PCM.
3
+ *
4
+ * The browser path gets its WAV from `convertAudioBufferToWav`, which needs a Web
5
+ * Audio `AudioBuffer`. Node has no such thing: the Kokoro model hands back a bare
6
+ * `Float32Array`, and long documents are synthesized one chunk at a time and
7
+ * joined before encoding. Both of those are pure buffer work, so they live here
8
+ * with no platform dependency.
9
+ */
10
+ /**
11
+ * Joins per-chunk sample buffers into one contiguous track.
12
+ *
13
+ * @param chunks Sample buffers in playback order.
14
+ * @param silenceSamples Samples of silence to insert between chunks, so the
15
+ * seams between separately synthesized chunks do not sound clipped.
16
+ */
17
+ export declare function concatSamples(chunks: Float32Array[], silenceSamples?: number): Float32Array;
18
+ /**
19
+ * Encodes Float32 samples (nominally -1..1) as a 16-bit PCM WAV file.
20
+ *
21
+ * @param samples Interleaved samples when `channels` is greater than 1.
22
+ * @param sampleRate Sample rate of the audio, e.g. 24000 for Kokoro.
23
+ * @param channels Channel count. Kokoro is mono.
24
+ */
25
+ export declare function encodeWav(samples: Float32Array, sampleRate?: number, channels?: number): ArrayBuffer;
26
+ /** Duration of a Float32 track in seconds, for progress and log lines. */
27
+ export declare function wavDurationSeconds(samples: Float32Array, sampleRate: number): number;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "use-voice-control",
3
- "version": "0.1.86",
4
- "description": "React voice control with speech transcription, vocalization, and interruption (STT/TTS/VAD) support.",
3
+ "version": "0.1.87",
4
+ "description": "React voice control with speech transcription, vocalization, and interruption (STT/TTS/VAD) support, plus a CLI that reads Markdown and text files aloud to audio files.",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/OpenSourceAGI/qwksearch-research-agent.git",
@@ -10,6 +10,9 @@
10
10
  "type": "module",
11
11
  "main": "./dist/index.js",
12
12
  "types": "./dist/index.d.ts",
13
+ "bin": {
14
+ "use-voice-control": "./bin/use-voice-control.mjs"
15
+ },
13
16
  "exports": {
14
17
  ".": {
15
18
  "types": "./dist/index.d.ts",
@@ -23,6 +26,14 @@
23
26
  "types": "./dist/react/index.d.ts",
24
27
  "import": "./dist/react.js"
25
28
  },
29
+ "./node": {
30
+ "types": "./dist/node/index.d.ts",
31
+ "import": "./dist/node.js"
32
+ },
33
+ "./markdown": {
34
+ "types": "./dist/utils/markdown-to-speech.d.ts",
35
+ "import": "./dist/markdown.js"
36
+ },
26
37
  "./api-client": {
27
38
  "types": "./speech/api-client.ts",
28
39
  "import": "./speech/api-client.ts"
@@ -31,11 +42,11 @@
31
42
  "files": [
32
43
  "dist",
33
44
  "speech",
45
+ "bin",
34
46
  "README.md"
35
47
  ],
36
48
  "scripts": {
37
49
  "build": "tsc && vite build",
38
- "prepare": "tsc && vite build",
39
50
  "dev": "vite build --watch",
40
51
  "type-check": "tsc --noEmit",
41
52
  "test": "vitest run",
@@ -44,9 +55,11 @@
44
55
  "dependencies": {
45
56
  "@huggingface/transformers": "^3.8.1",
46
57
  "lucide-react": "^0.344.0",
47
- "@moonshine-ai/moonshine-js": "^0.1.29"
58
+ "@moonshine-ai/moonshine-js": "^0.1.29",
59
+ "kokoro-js": "^1.2.1"
48
60
  },
49
61
  "devDependencies": {
62
+ "@types/node": "^22.10.2",
50
63
  "@types/react": "^19.2.17",
51
64
  "@vitejs/plugin-react": "^4.3.4",
52
65
  "@types/react-dom": "^19.2.3",
@@ -18,4 +18,14 @@ export {
18
18
  type TranscriberEngine,
19
19
  } from "./live-transcriber";
20
20
 
21
+ export {
22
+ looksLikeMarkdown,
23
+ markdownToSpeech,
24
+ markdownToSpeechSegments,
25
+ stripInlineMarkdown,
26
+ type MarkdownToSpeechOptions,
27
+ type SpeechSegment,
28
+ type SpeechSegmentType,
29
+ } from "../utils/markdown-to-speech";
30
+
21
31
  export type { TTSProvider, KokoroVoice, DeepgramSpeaker } from "../types/types";
@@ -13,6 +13,11 @@
13
13
  * `speechSynthesis` rather than failing, so the feature still works everywhere.
14
14
  */
15
15
  import type { TTSProvider } from "../types/types";
16
+ import {
17
+ looksLikeMarkdown,
18
+ markdownToSpeech,
19
+ type MarkdownToSpeechOptions,
20
+ } from "../utils/markdown-to-speech";
16
21
  import { splitTextSmart } from "../utils/semantic-split.js";
17
22
 
18
23
  export type ReadAloudState = "idle" | "loading" | "speaking" | "paused";
@@ -44,6 +49,14 @@ export interface ReadAloudOptions {
44
49
  * seams between chunks more audible. Default 240.
45
50
  */
46
51
  maxChunkLength?: number;
52
+ /**
53
+ * How to read the text handed to `speak()`. `auto` (default) converts text
54
+ * that looks like Markdown so "##" and "**" are not read out; `markdown`
55
+ * always converts; `text` never does.
56
+ */
57
+ format?: "auto" | "markdown" | "text";
58
+ /** Markdown conversion options, when the text is read as Markdown. */
59
+ markdown?: MarkdownToSpeechOptions;
47
60
  /** Override synthesis entirely (tests, a bring-your-own-TTS host app). */
48
61
  synthesize?: SynthesizeFn;
49
62
  /** Called as each chunk starts playing. */
@@ -102,7 +115,7 @@ export class ReadAloudController {
102
115
  this.stop();
103
116
 
104
117
  const maxChunkLength = this.options.maxChunkLength ?? DEFAULT_MAX_CHUNK;
105
- const chunks = splitTextSmart(text ?? "", maxChunkLength)
118
+ const chunks = splitTextSmart(this.toSpeakableText(text ?? ""), maxChunkLength)
106
119
  .map((chunk) => chunk.trim())
107
120
  .filter((chunk) => chunk.length > 0);
108
121
 
@@ -187,6 +200,19 @@ export class ReadAloudController {
187
200
  }
188
201
  }
189
202
 
203
+ /**
204
+ * Converts Markdown to spoken words before chunking, so a document read out
205
+ * of an editor does not have its syntax read back to the listener.
206
+ */
207
+ private toSpeakableText(text: string): string {
208
+ const format = this.options.format ?? "auto";
209
+ if (format === "text") return text;
210
+ if (format === "markdown" || looksLikeMarkdown(text)) {
211
+ return markdownToSpeech(text, this.options.markdown);
212
+ }
213
+ return text;
214
+ }
215
+
190
216
  private setState(state: ReadAloudState): void {
191
217
  if (this.state === state) return;
192
218
  this.state = state;
@@ -0,0 +1,164 @@
1
+ /**
2
+ * @fileoverview Kokoro TTS running locally on Node (or Bun/Deno) via `kokoro-js`.
3
+ *
4
+ * This is the engine behind `generateSpeech({ provider: "kokoro" })` on a server
5
+ * and behind the `use-voice-control` CLI. The model runs on the CPU through
6
+ * onnxruntime, so nothing is sent to a third-party service; the first call
7
+ * downloads the weights (~90 MB at the default `q8`) into the Hugging Face cache
8
+ * and every later call reuses the instance held in this module.
9
+ *
10
+ * Kokoro's context is a few hundred phonemes, so anything longer than a
11
+ * paragraph has to be synthesized in pieces. `synthesizeSamples` chunks on
12
+ * sentence and paragraph boundaries with `splitTextSmart`, runs the chunks in
13
+ * order, and joins the audio with a short silence at each seam.
14
+ */
15
+ import type { TTSResult } from "../types/types";
16
+ import { splitTextSmart } from "../utils/semantic-split.js";
17
+ import { concatSamples, encodeWav } from "../utils/wav";
18
+
19
+ export type KokoroDtype = "fp32" | "fp16" | "q8" | "q4" | "q4f16";
20
+ export type KokoroDevice = "wasm" | "webgpu" | "cpu";
21
+
22
+ export interface KokoroNodeOptions {
23
+ /** Voice id, e.g. `af_heart`. Default `af_heart`. */
24
+ voice?: string;
25
+ /** Speaking rate; 1 is the model's natural pace. Default 1. */
26
+ speed?: number;
27
+ /** Hugging Face model id. Default `onnx-community/Kokoro-82M-v1.0-ONNX`. */
28
+ model?: string;
29
+ /** Weight quantization. `q8` (default) is the best size/quality trade on CPU. */
30
+ dtype?: KokoroDtype;
31
+ /** Execution device. Default `cpu`. */
32
+ device?: KokoroDevice;
33
+ /**
34
+ * Target chunk size in characters. Kokoro's context is limited, so long text
35
+ * is split before synthesis; 400 leaves headroom for phoneme expansion.
36
+ */
37
+ maxChunkLength?: number;
38
+ /** Silence inserted between chunks, in milliseconds. Default 120. */
39
+ gapMs?: number;
40
+ /** Model download/loading progress, forwarded from transformers.js. */
41
+ onModelProgress?: (progress: unknown) => void;
42
+ /** Called before each chunk is synthesized, for CLI progress output. */
43
+ onChunk?: (info: { index: number; total: number; text: string }) => void;
44
+ }
45
+
46
+ export interface KokoroAudio {
47
+ samples: Float32Array;
48
+ sampleRate: number;
49
+ }
50
+
51
+ const DEFAULT_MODEL = "onnx-community/Kokoro-82M-v1.0-ONNX";
52
+ const DEFAULT_VOICE = "af_heart";
53
+ const DEFAULT_DTYPE: KokoroDtype = "q8";
54
+ const DEFAULT_DEVICE: KokoroDevice = "cpu";
55
+ const DEFAULT_MAX_CHUNK = 400;
56
+ const DEFAULT_GAP_MS = 120;
57
+
58
+ /** One loaded model per model/dtype/device combination, keyed by those three. */
59
+ const instances = new Map<string, Promise<any>>();
60
+
61
+ /** True on Node, Bun and Deno — anywhere `kokoro-js` can load onnxruntime-node. */
62
+ function isServerRuntime(): boolean {
63
+ const g = globalThis as any;
64
+ return Boolean(g.process?.versions?.node || g.Bun || g.Deno);
65
+ }
66
+
67
+ /**
68
+ * Loads (and caches) a `kokoro-js` instance.
69
+ *
70
+ * `kokoro-js` is an optional dependency: a browser-only consumer should not have
71
+ * to download onnxruntime-node, so a missing install is reported as an
72
+ * actionable error rather than a module-resolution stack trace.
73
+ */
74
+ export async function loadKokoroTTS(options: KokoroNodeOptions = {}): Promise<any> {
75
+ const model = options.model ?? DEFAULT_MODEL;
76
+ const dtype = options.dtype ?? DEFAULT_DTYPE;
77
+ const device = options.device ?? (isServerRuntime() ? DEFAULT_DEVICE : "wasm");
78
+ const key = `${model}|${dtype}|${device}`;
79
+
80
+ const cached = instances.get(key);
81
+ if (cached) return cached;
82
+
83
+ const loading = (async () => {
84
+ let KokoroTTS: any;
85
+ try {
86
+ // Kept in a variable so bundlers treat this as an optional runtime
87
+ // dependency rather than something to resolve at build time.
88
+ const specifier = "kokoro-js";
89
+ ({ KokoroTTS } = await import(/* @vite-ignore */ specifier));
90
+ } catch (error) {
91
+ throw new Error(
92
+ "Local Kokoro speech needs the optional `kokoro-js` package. " +
93
+ "Install it with `npm install kokoro-js`. " +
94
+ `(import failed: ${error instanceof Error ? error.message : String(error)})`
95
+ );
96
+ }
97
+
98
+ return KokoroTTS.from_pretrained(model, {
99
+ dtype,
100
+ device,
101
+ progress_callback: options.onModelProgress,
102
+ });
103
+ })();
104
+
105
+ instances.set(key, loading);
106
+
107
+ try {
108
+ return await loading;
109
+ } catch (error) {
110
+ // A failed download must not poison every later attempt.
111
+ instances.delete(key);
112
+ throw error;
113
+ }
114
+ }
115
+
116
+ /** Drops cached model instances. Mainly useful in tests and long-lived workers. */
117
+ export function resetKokoroCache(): void {
118
+ instances.clear();
119
+ }
120
+
121
+ /**
122
+ * Synthesizes `text` to raw samples, chunking anything longer than the model's
123
+ * comfortable context.
124
+ */
125
+ export async function synthesizeSamples(
126
+ text: string,
127
+ options: KokoroNodeOptions = {}
128
+ ): Promise<KokoroAudio> {
129
+ const trimmed = (text ?? "").trim();
130
+ if (!trimmed) throw new Error("Text is required");
131
+
132
+ const chunks = splitTextSmart(trimmed, options.maxChunkLength ?? DEFAULT_MAX_CHUNK)
133
+ .map((chunk: string) => chunk.trim())
134
+ .filter((chunk: string) => chunk.length > 0);
135
+
136
+ const tts = await loadKokoroTTS(options);
137
+ const voice = options.voice ?? DEFAULT_VOICE;
138
+ const speed = options.speed ?? 1;
139
+
140
+ const rendered: Float32Array[] = [];
141
+ let sampleRate = 24000;
142
+
143
+ for (let index = 0; index < chunks.length; index += 1) {
144
+ options.onChunk?.({ index, total: chunks.length, text: chunks[index] });
145
+ const audio = await tts.generate(chunks[index], { voice, speed });
146
+ rendered.push(audio.audio);
147
+ sampleRate = audio.sampling_rate ?? sampleRate;
148
+ }
149
+
150
+ const gapMs = options.gapMs ?? DEFAULT_GAP_MS;
151
+ return {
152
+ samples: concatSamples(rendered, Math.round((gapMs / 1000) * sampleRate)),
153
+ sampleRate,
154
+ };
155
+ }
156
+
157
+ /** Synthesizes `text` and encodes it as a 16-bit PCM WAV file. */
158
+ export async function synthesizeWav(
159
+ text: string,
160
+ options: KokoroNodeOptions = {}
161
+ ): Promise<TTSResult> {
162
+ const { samples, sampleRate } = await synthesizeSamples(text, options);
163
+ return { audio: encodeWav(samples, sampleRate), contentType: "audio/wav" };
164
+ }
@@ -1,81 +1,28 @@
1
1
  /**
2
- * @fileoverview Kokoro TTS provider implementation using Hugging Face transformers
3
- * Runs on Node.js CPU via transformers library
2
+ * @fileoverview Kokoro TTS provider for `generateSpeech`.
3
+ *
4
+ * The work happens in `kokoro-node.ts`, which runs the Kokoro model locally
5
+ * through `kokoro-js`; this file is the thin provider adapter that validates the
6
+ * voice and returns the package's `TTSResult` shape.
4
7
  */
5
8
  import type { TTSResult } from "../types/types";
6
9
  import { KOKORO_VOICES, type KokoroVoice } from "../types/types";
7
-
8
- let ttsInstance: any = null;
9
- let modelLoading: Promise<any> | null = null;
10
-
11
- /**
12
- * Lazy-load Kokoro model using Hugging Face transformers (happens once per server instance)
13
- */
14
- async function getKokoroTTS() {
15
- if (ttsInstance) return ttsInstance;
16
-
17
- if (modelLoading) {
18
- await modelLoading;
19
- return ttsInstance;
20
- }
21
-
22
- modelLoading = (async () => {
23
- try {
24
- // Dynamic import to load transformers library. The specifier is kept in
25
- // a variable so the type-checker/bundler treats it as an optional runtime
26
- // dependency (see optionalDependencies) rather than a build-time one.
27
- const transformersModule = "@huggingface/transformers";
28
- const transformers: any = await import(/* @vite-ignore */ transformersModule);
29
- const { StyleTextToSpeech2Model, AutoTokenizer } = transformers;
30
-
31
- const model_id = "hexgrad/Kokoro-82M";
32
-
33
- // Load model and tokenizer in parallel
34
- const [model, tokenizer] = await Promise.all([
35
- StyleTextToSpeech2Model.from_pretrained(model_id, {
36
- device: "cpu",
37
- dtype: "q8"
38
- }),
39
- AutoTokenizer.from_pretrained(model_id)
40
- ]);
41
-
42
- ttsInstance = { model, tokenizer };
43
-
44
- console.log("[Kokoro] Model loaded successfully");
45
- } catch (error) {
46
- console.error("[Kokoro] Failed to load model:", error);
47
- throw error;
48
- } finally {
49
- modelLoading = null;
50
- }
51
- })();
52
-
53
- await modelLoading;
54
- return ttsInstance;
55
- }
10
+ import { synthesizeWav, type KokoroNodeOptions } from "./kokoro-node";
56
11
 
57
12
  /**
58
- * Generate speech from text using Kokoro
13
+ * Generate speech from text using Kokoro.
14
+ *
15
+ * Long text is chunked and joined automatically. An unrecognised voice falls
16
+ * back to `af_heart` rather than failing the request.
59
17
  */
60
18
  export async function generateKokoroSpeech(
61
19
  text: string,
62
- voice: string = "af_heart"
20
+ voice: string = "af_heart",
21
+ options: Omit<KokoroNodeOptions, "voice"> = {}
63
22
  ): Promise<TTSResult> {
64
- // Validate voice
65
23
  const kokoroVoice = KOKORO_VOICES.includes(voice as KokoroVoice)
66
24
  ? (voice as KokoroVoice)
67
25
  : "af_heart";
68
26
 
69
- const tts = await getKokoroTTS();
70
-
71
- try {
72
- // For Node.js server-side TTS, we'd need the complete transformers implementation
73
- // This is a placeholder showing the expected interface
74
- // Consider using the browser-based implementation via Web Workers for production
75
-
76
- throw new Error("Server-side Kokoro TTS requires additional setup. Use the browser-based implementation via Web Workers.");
77
- } catch (error) {
78
- console.error("[Kokoro] Error generating speech:", error);
79
- throw error;
80
- }
27
+ return synthesizeWav(text, { ...options, voice: kokoroVoice });
81
28
  }
package/speech/index.ts CHANGED
@@ -10,6 +10,20 @@ import { generateDeepgramSpeech } from "./core/deepgram";
10
10
 
11
11
  export * from "./types/types";
12
12
 
13
+ // Markdown handling is exported from the root entry so a server route can turn a
14
+ // document into speakable text with the same rules the CLI and the browser use.
15
+ export {
16
+ looksLikeMarkdown,
17
+ markdownToSpeech,
18
+ markdownToSpeechSegments,
19
+ stripInlineMarkdown,
20
+ type MarkdownToSpeechOptions,
21
+ type SpeechSegment,
22
+ type SpeechSegmentType,
23
+ } from "./utils/markdown-to-speech";
24
+
25
+ export { encodeWav, concatSamples, wavDurationSeconds } from "./utils/wav";
26
+
13
27
  /**
14
28
  * Generate speech from text using the specified provider
15
29
  *