use-voice-control 0.1.85 → 0.1.87
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +206 -21
- package/bin/use-voice-control.mjs +20 -0
- package/dist/cli.js +374 -0
- package/dist/cli.js.map +1 -0
- package/dist/client/index.d.ts +1 -0
- package/dist/client/read-aloud.d.ts +14 -0
- package/dist/client.js +132 -157
- package/dist/client.js.map +1 -1
- package/dist/core/kokoro-node.d.ts +51 -0
- package/dist/core/kokoro.d.ts +6 -2
- package/dist/index.d.ts +2 -0
- package/dist/index.js +33 -76
- package/dist/index.js.map +1 -1
- package/dist/kokoro-node-DJ_Rxp_N.js +144 -0
- package/dist/kokoro-node-DJ_Rxp_N.js.map +1 -0
- package/dist/markdown.js +213 -0
- package/dist/markdown.js.map +1 -0
- package/dist/node/cli.d.ts +52 -0
- package/dist/node/document.d.ts +45 -0
- package/dist/node/index.d.ts +18 -0
- package/dist/node/render.d.ts +30 -0
- package/dist/node.js +27 -0
- package/dist/node.js.map +1 -0
- package/dist/react.js +13 -11
- package/dist/react.js.map +1 -1
- package/dist/semantic-split-CXhk-k1F.js +44 -0
- package/dist/semantic-split-CXhk-k1F.js.map +1 -0
- package/dist/types/types.d.ts +14 -1
- package/dist/utils/markdown-to-speech.d.ts +76 -0
- package/dist/utils/wav.d.ts +27 -0
- package/package.json +17 -4
- package/speech/client/index.ts +10 -0
- package/speech/client/read-aloud.ts +27 -1
- package/speech/core/kokoro-node.ts +164 -0
- package/speech/core/kokoro.ts +13 -66
- package/speech/index.ts +14 -0
- package/speech/node/cli.ts +509 -0
- package/speech/node/document.ts +126 -0
- package/speech/node/index.ts +64 -0
- package/speech/node/render.ts +79 -0
- package/speech/react/useReadAloud.ts +2 -0
- package/speech/types/types.ts +34 -5
- package/speech/utils/markdown-to-speech.ts +426 -0
- package/speech/utils/wav.ts +89 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Rendering a loaded document to an audio file on disk.
|
|
3
|
+
*
|
|
4
|
+
* Sits between `document.ts` (what to say) and `core/kokoro-node.ts` (how to say
|
|
5
|
+
* it): resolves the output path, runs the synthesizer, and writes the WAV.
|
|
6
|
+
*/
|
|
7
|
+
import {
|
|
8
|
+
synthesizeSamples,
|
|
9
|
+
type KokoroAudio,
|
|
10
|
+
type KokoroNodeOptions,
|
|
11
|
+
} from "../core/kokoro-node";
|
|
12
|
+
import { encodeWav, wavDurationSeconds } from "../utils/wav";
|
|
13
|
+
import { extensionOf, loadDocument, type LoadDocumentOptions } from "./document";
|
|
14
|
+
|
|
15
|
+
export interface RenderOptions extends LoadDocumentOptions, KokoroNodeOptions {
|
|
16
|
+
/** Where to write the audio. `-` writes the WAV to stdout. */
|
|
17
|
+
output?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Replace the synthesizer. Used by the tests, and by hosts that already have a
|
|
20
|
+
* model loaded and do not want a second copy.
|
|
21
|
+
*/
|
|
22
|
+
synthesize?: (text: string, options: KokoroNodeOptions) => Promise<KokoroAudio>;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export interface RenderResult {
|
|
26
|
+
/** Path written, or `-` when the audio went to stdout. */
|
|
27
|
+
output: string;
|
|
28
|
+
/** The text that was spoken, after Markdown conversion. */
|
|
29
|
+
text: string;
|
|
30
|
+
format: "markdown" | "text";
|
|
31
|
+
source: string;
|
|
32
|
+
durationSeconds: number;
|
|
33
|
+
bytes: number;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** `notes.md` becomes `notes.wav`, next to the original. */
|
|
37
|
+
export function defaultOutputPath(input: string): string {
|
|
38
|
+
const extension = extensionOf(input);
|
|
39
|
+
const base = extension ? input.slice(0, -extension.length) : input;
|
|
40
|
+
return `${base}.wav`;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Reads a document, synthesizes it, and writes a WAV file.
|
|
45
|
+
*
|
|
46
|
+
* Returns the spoken text alongside the file details so a caller can show what
|
|
47
|
+
* was read without converting the document a second time.
|
|
48
|
+
*/
|
|
49
|
+
export async function renderDocument(options: RenderOptions): Promise<RenderResult> {
|
|
50
|
+
const document = await loadDocument(options);
|
|
51
|
+
|
|
52
|
+
if (!document.text.trim()) {
|
|
53
|
+
throw new Error(`Nothing to speak: ${document.source} has no readable text`);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const synthesize = options.synthesize ?? synthesizeSamples;
|
|
57
|
+
const { samples, sampleRate } = await synthesize(document.text, options);
|
|
58
|
+
const audio = encodeWav(samples, sampleRate);
|
|
59
|
+
|
|
60
|
+
const output =
|
|
61
|
+
options.output ??
|
|
62
|
+
(options.file && options.file !== "-" ? defaultOutputPath(options.file) : "out.wav");
|
|
63
|
+
|
|
64
|
+
if (output === "-") {
|
|
65
|
+
process.stdout.write(Buffer.from(audio));
|
|
66
|
+
} else {
|
|
67
|
+
const { writeFile } = await import("node:fs/promises");
|
|
68
|
+
await writeFile(output, Buffer.from(audio));
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
return {
|
|
72
|
+
output,
|
|
73
|
+
text: document.text,
|
|
74
|
+
format: document.format,
|
|
75
|
+
source: document.source,
|
|
76
|
+
durationSeconds: wavDurationSeconds(samples, sampleRate),
|
|
77
|
+
bytes: audio.byteLength,
|
|
78
|
+
};
|
|
79
|
+
}
|
package/speech/types/types.ts
CHANGED
|
@@ -15,16 +15,45 @@ export interface TTSResult {
|
|
|
15
15
|
contentType: string;
|
|
16
16
|
}
|
|
17
17
|
|
|
18
|
-
// Kokoro voices from the model
|
|
18
|
+
// Kokoro voices from the model. The `a`/`b` prefix is the accent (American /
|
|
19
|
+
// British English) and the `f`/`m` that follows it is the voice's gender, which
|
|
20
|
+
// is how `describeKokoroVoice` derives a listing without a second catalog.
|
|
19
21
|
export const KOKORO_VOICES = [
|
|
20
|
-
"af_heart", "af_alloy", "af_aoede", "af_bella",
|
|
21
|
-
"
|
|
22
|
-
"am_adam", "am_echo", "
|
|
23
|
-
"
|
|
22
|
+
"af_heart", "af_alloy", "af_aoede", "af_bella", "af_jessica",
|
|
23
|
+
"af_kore", "af_nicole", "af_nova", "af_river", "af_sarah", "af_sky",
|
|
24
|
+
"am_adam", "am_echo", "am_eric", "am_fenrir", "am_liam",
|
|
25
|
+
"am_michael", "am_onyx", "am_puck", "am_santa",
|
|
26
|
+
"bf_alice", "bf_emma", "bf_isabella", "bf_lily",
|
|
27
|
+
"bm_daniel", "bm_fable", "bm_george", "bm_lewis",
|
|
24
28
|
] as const;
|
|
25
29
|
|
|
26
30
|
export type KokoroVoice = (typeof KOKORO_VOICES)[number];
|
|
27
31
|
|
|
32
|
+
export interface KokoroVoiceDescription {
|
|
33
|
+
id: string;
|
|
34
|
+
/** Display name, e.g. `Heart` for `af_heart`. */
|
|
35
|
+
name: string;
|
|
36
|
+
/** `American English` or `British English`. */
|
|
37
|
+
accent: string;
|
|
38
|
+
gender: "Female" | "Male";
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Describes a voice from its id, so listings (`--list-voices`, a voice picker)
|
|
43
|
+
* do not need a hand-maintained table that can drift from `KOKORO_VOICES`.
|
|
44
|
+
*/
|
|
45
|
+
export function describeKokoroVoice(id: string): KokoroVoiceDescription {
|
|
46
|
+
const [accentCode, genderCode] = id;
|
|
47
|
+
const suffix = id.slice(3) || id;
|
|
48
|
+
|
|
49
|
+
return {
|
|
50
|
+
id,
|
|
51
|
+
name: suffix.charAt(0).toUpperCase() + suffix.slice(1),
|
|
52
|
+
accent: accentCode === "b" ? "British English" : "American English",
|
|
53
|
+
gender: genderCode === "m" ? "Male" : "Female",
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
|
|
28
57
|
// Deepgram Aura speakers
|
|
29
58
|
export const DEEPGRAM_SPEAKERS = [
|
|
30
59
|
"angus", "asteria", "arcas", "orion", "orpheus", "athena",
|
|
@@ -0,0 +1,426 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Turns Markdown into text a speech synthesizer can read.
|
|
3
|
+
*
|
|
4
|
+
* Markdown handed straight to a TTS engine is read literally: "hash hash Getting
|
|
5
|
+
* started", "star star important star star". This module parses the document
|
|
6
|
+
* instead and emits only the words, keeping the *structure* the marks encoded —
|
|
7
|
+
* a heading becomes its own spoken segment ending in a full stop so the
|
|
8
|
+
* synthesizer pauses, a list item becomes one sentence, a fenced code block is
|
|
9
|
+
* announced rather than spelled out.
|
|
10
|
+
*
|
|
11
|
+
* It is deliberately dependency-free and runs in the browser as well as Node, so
|
|
12
|
+
* `ReadAloudController` and the CLI can share it.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/** Kind of block the converter recognised. */
|
|
16
|
+
export type SpeechSegmentType =
|
|
17
|
+
| "heading"
|
|
18
|
+
| "paragraph"
|
|
19
|
+
| "list-item"
|
|
20
|
+
| "quote"
|
|
21
|
+
| "code"
|
|
22
|
+
| "table-row";
|
|
23
|
+
|
|
24
|
+
export interface SpeechSegment {
|
|
25
|
+
type: SpeechSegmentType;
|
|
26
|
+
/** Spoken text of the block, with every Markdown mark already removed. */
|
|
27
|
+
text: string;
|
|
28
|
+
/** Heading level 1-6. Only set on `heading` segments. */
|
|
29
|
+
level?: number;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface MarkdownToSpeechOptions {
|
|
33
|
+
/**
|
|
34
|
+
* How headings are spoken. `text` (default) reads the heading words on their
|
|
35
|
+
* own so they land between pauses; `announce` prefixes them with "Heading:";
|
|
36
|
+
* `skip` drops them entirely.
|
|
37
|
+
*/
|
|
38
|
+
headings?: "text" | "announce" | "skip";
|
|
39
|
+
/**
|
|
40
|
+
* What to do with fenced code blocks. `announce` (default) replaces the block
|
|
41
|
+
* with a short spoken note, `read` reads the code verbatim, `skip` drops it.
|
|
42
|
+
*/
|
|
43
|
+
codeBlocks?: "announce" | "read" | "skip";
|
|
44
|
+
/** `text` (default) speaks the link text and drops the URL; `text-and-url` reads both. */
|
|
45
|
+
links?: "text" | "text-and-url";
|
|
46
|
+
/** `alt` (default) speaks the image's alt text; `skip` drops images. */
|
|
47
|
+
images?: "alt" | "skip";
|
|
48
|
+
/** Read the YAML front matter block. Off by default. */
|
|
49
|
+
frontMatter?: boolean;
|
|
50
|
+
/** `rows` (default) reads table rows as comma-separated cells; `skip` drops tables. */
|
|
51
|
+
tables?: "rows" | "skip";
|
|
52
|
+
/**
|
|
53
|
+
* Append a full stop to blocks that do not end in punctuation, so the
|
|
54
|
+
* synthesizer pauses between them instead of running them together.
|
|
55
|
+
* Default true.
|
|
56
|
+
*/
|
|
57
|
+
addTerminalPunctuation?: boolean;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
type ResolvedOptions = Required<MarkdownToSpeechOptions>;
|
|
61
|
+
|
|
62
|
+
const DEFAULTS: ResolvedOptions = {
|
|
63
|
+
headings: "text",
|
|
64
|
+
codeBlocks: "announce",
|
|
65
|
+
links: "text",
|
|
66
|
+
images: "alt",
|
|
67
|
+
frontMatter: false,
|
|
68
|
+
tables: "rows",
|
|
69
|
+
addTerminalPunctuation: true,
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
/** Spoken names for code fence languages that would otherwise be read as letters. */
|
|
73
|
+
const LANGUAGE_NAMES: Record<string, string> = {
|
|
74
|
+
js: "JavaScript",
|
|
75
|
+
jsx: "JavaScript",
|
|
76
|
+
mjs: "JavaScript",
|
|
77
|
+
cjs: "JavaScript",
|
|
78
|
+
ts: "TypeScript",
|
|
79
|
+
tsx: "TypeScript",
|
|
80
|
+
py: "Python",
|
|
81
|
+
rb: "Ruby",
|
|
82
|
+
rs: "Rust",
|
|
83
|
+
sh: "shell",
|
|
84
|
+
bash: "shell",
|
|
85
|
+
zsh: "shell",
|
|
86
|
+
shell: "shell",
|
|
87
|
+
console: "shell",
|
|
88
|
+
yml: "YAML",
|
|
89
|
+
yaml: "YAML",
|
|
90
|
+
md: "Markdown",
|
|
91
|
+
json: "JSON",
|
|
92
|
+
html: "HTML",
|
|
93
|
+
css: "CSS",
|
|
94
|
+
sql: "SQL",
|
|
95
|
+
go: "Go",
|
|
96
|
+
java: "Java",
|
|
97
|
+
c: "C",
|
|
98
|
+
cpp: "C plus plus",
|
|
99
|
+
cs: "C sharp",
|
|
100
|
+
php: "PHP",
|
|
101
|
+
swift: "Swift",
|
|
102
|
+
kt: "Kotlin",
|
|
103
|
+
diff: "diff",
|
|
104
|
+
text: "",
|
|
105
|
+
txt: "",
|
|
106
|
+
plaintext: "",
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
const ENTITIES: Record<string, string> = {
|
|
110
|
+
amp: "&",
|
|
111
|
+
lt: "<",
|
|
112
|
+
gt: ">",
|
|
113
|
+
quot: '"',
|
|
114
|
+
apos: "'",
|
|
115
|
+
nbsp: " ",
|
|
116
|
+
mdash: "—",
|
|
117
|
+
ndash: "–",
|
|
118
|
+
hellip: "…",
|
|
119
|
+
ldquo: '"',
|
|
120
|
+
rdquo: '"',
|
|
121
|
+
lsquo: "'",
|
|
122
|
+
rsquo: "'",
|
|
123
|
+
};
|
|
124
|
+
|
|
125
|
+
const ATX_HEADING = /^ {0,3}(#{1,6})(?:\s+(.*?))?\s*$/;
|
|
126
|
+
const SETEXT_UNDERLINE = /^ {0,3}(=+|-+)\s*$/;
|
|
127
|
+
const THEMATIC_BREAK = /^ {0,3}([-*_])[ \t]*(?:\1[ \t]*){2,}$/;
|
|
128
|
+
const FENCE_OPEN = /^ {0,3}(`{3,}|~{3,})\s*([^`]*)$/;
|
|
129
|
+
const LIST_ITEM = /^ *(?:[-+*]|\d{1,9}[.)])(?:\s+(.*))?$/;
|
|
130
|
+
const ORDERED_ITEM = /^ *(\d{1,9})[.)]\s+(.*)$/;
|
|
131
|
+
const TASK_MARKER = /^\[([ xX])\]\s+/;
|
|
132
|
+
const LINK_REFERENCE_DEFINITION = /^ {0,3}\[[^\]]+\]:\s*\S+.*$/;
|
|
133
|
+
const TABLE_DELIMITER_ROW = /^[\s|:-]*-[\s|:-]*$/;
|
|
134
|
+
|
|
135
|
+
/** Marks the slot where a protected substring was lifted out of the line. */
|
|
136
|
+
const PLACEHOLDER_OPEN = "\u0000";
|
|
137
|
+
const PLACEHOLDER_CLOSE = "\u0001";
|
|
138
|
+
const PLACEHOLDER_PATTERN = /\u0000(\d+)\u0001/g;
|
|
139
|
+
|
|
140
|
+
/** Ends a block with a full stop unless it already ends in sentence punctuation. */
|
|
141
|
+
function withTerminalPunctuation(text: string): string {
|
|
142
|
+
return /[.!?:;,…]["')\]]?$/.test(text) ? text : `${text}.`;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Replaces `&`-style entities with the characters they stand for. */
|
|
146
|
+
function decodeEntities(text: string): string {
|
|
147
|
+
return text
|
|
148
|
+
.replace(/&#(\d{1,7});/g, (_m, code) => String.fromCodePoint(Number(code)))
|
|
149
|
+
.replace(/&#[xX]([0-9a-fA-F]{1,6});/g, (_m, code) =>
|
|
150
|
+
String.fromCodePoint(parseInt(code, 16))
|
|
151
|
+
)
|
|
152
|
+
.replace(/&([a-zA-Z]+);/g, (match, name) => ENTITIES[name.toLowerCase()] ?? match);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Strips the inline marks — emphasis, code spans, links, images, raw HTML — from
|
|
157
|
+
* a single line, leaving the words behind.
|
|
158
|
+
*/
|
|
159
|
+
export function stripInlineMarkdown(
|
|
160
|
+
input: string,
|
|
161
|
+
options: MarkdownToSpeechOptions = {}
|
|
162
|
+
): string {
|
|
163
|
+
const opts: ResolvedOptions = { ...DEFAULTS, ...options };
|
|
164
|
+
const protectedRuns: string[] = [];
|
|
165
|
+
const hold = (value: string): string =>
|
|
166
|
+
`${PLACEHOLDER_OPEN}${protectedRuns.push(value) - 1}${PLACEHOLDER_CLOSE}`;
|
|
167
|
+
|
|
168
|
+
let text = input;
|
|
169
|
+
|
|
170
|
+
// Backslash escapes and code spans come out first: whatever they contain must
|
|
171
|
+
// survive the emphasis pass untouched.
|
|
172
|
+
text = text.replace(/\\([\\`*_{}[\]()#+\-.!>~|])/g, (_m, char) => hold(char));
|
|
173
|
+
text = text.replace(/(`+)([^`]|[^`][\s\S]*?[^`])\1(?!`)/g, (_m, _ticks, code) =>
|
|
174
|
+
hold(String(code).trim())
|
|
175
|
+
);
|
|
176
|
+
|
|
177
|
+
text = text.replace(/<!--[\s\S]*?-->/g, "");
|
|
178
|
+
|
|
179
|
+
// Images before links: `` is a link whose text starts with `!`.
|
|
180
|
+
text = text.replace(/!\[([^\]]*)\]\([^)]*\)/g, (_m, alt) =>
|
|
181
|
+
opts.images === "skip" ? "" : String(alt)
|
|
182
|
+
);
|
|
183
|
+
text = text.replace(/!\[([^\]]*)\]\[[^\]]*\]/g, (_m, alt) =>
|
|
184
|
+
opts.images === "skip" ? "" : String(alt)
|
|
185
|
+
);
|
|
186
|
+
|
|
187
|
+
// Footnote references carry no spoken meaning.
|
|
188
|
+
text = text.replace(/\[\^[^\]]+\]/g, "");
|
|
189
|
+
|
|
190
|
+
text = text.replace(/\[([^\]]*)\]\(\s*<?([^\s)]*)>?[^)]*\)/g, (_m, label, url) =>
|
|
191
|
+
opts.links === "text-and-url" && url ? `${label}, ${url}` : String(label)
|
|
192
|
+
);
|
|
193
|
+
text = text.replace(/\[([^\]]*)\]\[[^\]]*\]/g, "$1");
|
|
194
|
+
|
|
195
|
+
// Autolinks: `<https://example.com>` reads better as the bare address.
|
|
196
|
+
text = text.replace(/<((?:https?|mailto):[^>\s]+)>/gi, (_m, url) =>
|
|
197
|
+
opts.links === "text-and-url"
|
|
198
|
+
? String(url)
|
|
199
|
+
: String(url).replace(/^mailto:/i, "")
|
|
200
|
+
);
|
|
201
|
+
|
|
202
|
+
// Remaining angle brackets are raw HTML tags.
|
|
203
|
+
text = text.replace(/<\/?[a-zA-Z][^>]*>/g, "");
|
|
204
|
+
|
|
205
|
+
text = text.replace(/\*\*\*(?=\S)([\s\S]*?\S)\*\*\*/g, "$1");
|
|
206
|
+
text = text.replace(/\*\*(?=\S)([\s\S]*?\S)\*\*/g, "$1");
|
|
207
|
+
text = text.replace(/(?<![A-Za-z0-9_])__(?=\S)([\s\S]*?\S)__(?![A-Za-z0-9_])/g, "$1");
|
|
208
|
+
text = text.replace(/~~(?=\S)([\s\S]*?\S)~~/g, "$1");
|
|
209
|
+
text = text.replace(/\*(?=\S)([^*\n]*?\S)\*/g, "$1");
|
|
210
|
+
// Underscores are only emphasis between word boundaries, so `snake_case` survives.
|
|
211
|
+
text = text.replace(/(?<![A-Za-z0-9_])_(?=\S)([^_\n]*?\S)_(?![A-Za-z0-9_])/g, "$1");
|
|
212
|
+
|
|
213
|
+
text = decodeEntities(text);
|
|
214
|
+
text = text.replace(
|
|
215
|
+
PLACEHOLDER_PATTERN,
|
|
216
|
+
(_m, index) => protectedRuns[Number(index)] ?? ""
|
|
217
|
+
);
|
|
218
|
+
|
|
219
|
+
return text.replace(/[ \t]+/g, " ").trim();
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/** Spoken stand-in for a fenced code block, e.g. "TypeScript code block." */
|
|
223
|
+
function announceCode(language: string): string {
|
|
224
|
+
const key = language.trim().split(/[\s,{]/)[0]?.toLowerCase() ?? "";
|
|
225
|
+
const name = key in LANGUAGE_NAMES ? LANGUAGE_NAMES[key] : key;
|
|
226
|
+
return name ? `${name} code block.` : "Code block.";
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Parses Markdown into the blocks that should be spoken, in reading order.
|
|
231
|
+
*
|
|
232
|
+
* Use this when the caller wants the structure — to highlight the current
|
|
233
|
+
* heading, say, or to skip to a section. Callers that only need something to
|
|
234
|
+
* feed a synthesizer want {@link markdownToSpeech}.
|
|
235
|
+
*/
|
|
236
|
+
export function markdownToSpeechSegments(
|
|
237
|
+
markdown: string,
|
|
238
|
+
options: MarkdownToSpeechOptions = {}
|
|
239
|
+
): SpeechSegment[] {
|
|
240
|
+
const opts: ResolvedOptions = { ...DEFAULTS, ...options };
|
|
241
|
+
const segments: SpeechSegment[] = [];
|
|
242
|
+
|
|
243
|
+
let source = (markdown ?? "").replace(/\r\n?/g, "\n").replace(/\t/g, " ");
|
|
244
|
+
source = source.replace(/<!--[\s\S]*?-->/g, "");
|
|
245
|
+
|
|
246
|
+
const lines = source.split("\n");
|
|
247
|
+
let index = 0;
|
|
248
|
+
|
|
249
|
+
// YAML front matter: only when it opens on the very first line.
|
|
250
|
+
if (!opts.frontMatter && /^---\s*$/.test(lines[0] ?? "")) {
|
|
251
|
+
const closing = lines.findIndex((line, i) => i > 0 && /^(---|\.\.\.)\s*$/.test(line));
|
|
252
|
+
if (closing > 0) index = closing + 1;
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/** Soft-wrapped lines of the paragraph being accumulated. */
|
|
256
|
+
let paragraph: string[] = [];
|
|
257
|
+
let quoting = false;
|
|
258
|
+
|
|
259
|
+
const push = (type: SpeechSegmentType, text: string, level?: number): void => {
|
|
260
|
+
const trimmed = text.trim();
|
|
261
|
+
if (!trimmed) return;
|
|
262
|
+
const spoken = opts.addTerminalPunctuation ? withTerminalPunctuation(trimmed) : trimmed;
|
|
263
|
+
segments.push(level === undefined ? { type, text: spoken } : { type, text: spoken, level });
|
|
264
|
+
};
|
|
265
|
+
|
|
266
|
+
const flushParagraph = (): void => {
|
|
267
|
+
if (paragraph.length === 0) return;
|
|
268
|
+
const text = stripInlineMarkdown(paragraph.join(" "), opts);
|
|
269
|
+
paragraph = [];
|
|
270
|
+
const type = quoting ? "quote" : "paragraph";
|
|
271
|
+
quoting = false;
|
|
272
|
+
push(type, text);
|
|
273
|
+
};
|
|
274
|
+
|
|
275
|
+
for (; index < lines.length; index += 1) {
|
|
276
|
+
let line = lines[index];
|
|
277
|
+
|
|
278
|
+
// A fence closes any paragraph before it and swallows lines until it ends.
|
|
279
|
+
const fence = FENCE_OPEN.exec(line);
|
|
280
|
+
if (fence) {
|
|
281
|
+
flushParagraph();
|
|
282
|
+
const marker = fence[1];
|
|
283
|
+
const language = fence[2] ?? "";
|
|
284
|
+
const closingFence = new RegExp(`^ {0,3}\\${marker[0]}{${marker.length},}\\s*$`);
|
|
285
|
+
const body: string[] = [];
|
|
286
|
+
index += 1;
|
|
287
|
+
for (; index < lines.length; index += 1) {
|
|
288
|
+
if (closingFence.test(lines[index])) break;
|
|
289
|
+
body.push(lines[index]);
|
|
290
|
+
}
|
|
291
|
+
if (opts.codeBlocks === "announce") {
|
|
292
|
+
push("code", announceCode(language));
|
|
293
|
+
} else if (opts.codeBlocks === "read") {
|
|
294
|
+
push("code", body.join(" ").replace(/\s+/g, " "));
|
|
295
|
+
}
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
if (line.trim() === "") {
|
|
300
|
+
flushParagraph();
|
|
301
|
+
continue;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// Blockquote markers are stripped so the quoted content is parsed normally.
|
|
305
|
+
let isQuote = false;
|
|
306
|
+
while (/^ {0,3}>\s?/.test(line)) {
|
|
307
|
+
line = line.replace(/^ {0,3}>\s?/, "");
|
|
308
|
+
isQuote = true;
|
|
309
|
+
}
|
|
310
|
+
if (isQuote) {
|
|
311
|
+
if (paragraph.length === 0) quoting = true;
|
|
312
|
+
if (line.trim() === "") {
|
|
313
|
+
flushParagraph();
|
|
314
|
+
continue;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
if (THEMATIC_BREAK.test(line)) {
|
|
319
|
+
flushParagraph();
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
const heading = ATX_HEADING.exec(line);
|
|
324
|
+
if (heading) {
|
|
325
|
+
flushParagraph();
|
|
326
|
+
if (opts.headings === "skip") continue;
|
|
327
|
+
// A closing run of hashes (`## Title ##`) is decoration, not content.
|
|
328
|
+
const text = stripInlineMarkdown((heading[2] ?? "").replace(/\s+#+$/, ""), opts);
|
|
329
|
+
if (!text) continue;
|
|
330
|
+
push("heading", opts.headings === "announce" ? `Heading: ${text}` : text, heading[1].length);
|
|
331
|
+
continue;
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// Setext heading: the underline applies to the single line above it.
|
|
335
|
+
const underline = SETEXT_UNDERLINE.exec(line);
|
|
336
|
+
if (underline && paragraph.length > 0 && opts.headings !== "skip") {
|
|
337
|
+
const text = stripInlineMarkdown(paragraph.join(" "), opts);
|
|
338
|
+
paragraph = [];
|
|
339
|
+
quoting = false;
|
|
340
|
+
if (text) {
|
|
341
|
+
push(
|
|
342
|
+
"heading",
|
|
343
|
+
opts.headings === "announce" ? `Heading: ${text}` : text,
|
|
344
|
+
underline[1].startsWith("=") ? 1 : 2
|
|
345
|
+
);
|
|
346
|
+
}
|
|
347
|
+
continue;
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
if (LINK_REFERENCE_DEFINITION.test(line) && paragraph.length === 0) continue;
|
|
351
|
+
|
|
352
|
+
if (line.trimStart().startsWith("|")) {
|
|
353
|
+
flushParagraph();
|
|
354
|
+
if (opts.tables === "skip" || TABLE_DELIMITER_ROW.test(line)) continue;
|
|
355
|
+
const cells = line
|
|
356
|
+
.trim()
|
|
357
|
+
.replace(/^\|/, "")
|
|
358
|
+
.replace(/\|$/, "")
|
|
359
|
+
.split(/(?<!\\)\|/)
|
|
360
|
+
.map((cell) => stripInlineMarkdown(cell, opts))
|
|
361
|
+
.filter((cell) => cell.length > 0);
|
|
362
|
+
if (cells.length > 0) push("table-row", cells.join(", "));
|
|
363
|
+
continue;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
const item = LIST_ITEM.exec(line);
|
|
367
|
+
if (item) {
|
|
368
|
+
flushParagraph();
|
|
369
|
+
const ordered = ORDERED_ITEM.exec(line);
|
|
370
|
+
// Keep the number — a listener needs it — but drop bullets and checkboxes.
|
|
371
|
+
const body = (ordered ? ordered[2] : (item[1] ?? "")).replace(TASK_MARKER, "");
|
|
372
|
+
const text = stripInlineMarkdown(body, opts);
|
|
373
|
+
if (!text) continue;
|
|
374
|
+
push("list-item", ordered ? `${ordered[1]}. ${text}` : text);
|
|
375
|
+
continue;
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
paragraph.push(line.trim());
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
flushParagraph();
|
|
382
|
+
return segments;
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
/**
|
|
386
|
+
* Converts Markdown to plain text ready for a speech synthesizer.
|
|
387
|
+
*
|
|
388
|
+
* Blocks are separated by blank lines so downstream chunkers (`splitTextSmart`)
|
|
389
|
+
* break on them; runs of list items and table rows stay in one block so a list
|
|
390
|
+
* is not chopped into one request per bullet.
|
|
391
|
+
*/
|
|
392
|
+
export function markdownToSpeech(
|
|
393
|
+
markdown: string,
|
|
394
|
+
options: MarkdownToSpeechOptions = {}
|
|
395
|
+
): string {
|
|
396
|
+
const segments = markdownToSpeechSegments(markdown, options);
|
|
397
|
+
|
|
398
|
+
return segments.reduce((out, segment, i) => {
|
|
399
|
+
if (i === 0) return segment.text;
|
|
400
|
+
const previous = segments[i - 1].type;
|
|
401
|
+
const runsOn =
|
|
402
|
+
(segment.type === "list-item" || segment.type === "table-row") &&
|
|
403
|
+
previous === segment.type;
|
|
404
|
+
return `${out}${runsOn ? "\n" : "\n\n"}${segment.text}`;
|
|
405
|
+
}, "");
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
/**
|
|
409
|
+
* Guesses whether a blob of text is Markdown, for callers with no filename to go
|
|
410
|
+
* on (piped stdin, a paste). Errs towards `false`: prose read as Markdown is
|
|
411
|
+
* mostly unchanged anyway, so only reasonably clear signals count.
|
|
412
|
+
*/
|
|
413
|
+
export function looksLikeMarkdown(text: string): boolean {
|
|
414
|
+
if (!text) return false;
|
|
415
|
+
const sample = text.slice(0, 20000);
|
|
416
|
+
const signals = [
|
|
417
|
+
/^ {0,3}#{1,6}\s+\S/m,
|
|
418
|
+
/^ {0,3}(?:```|~~~)/m,
|
|
419
|
+
/^ {0,3}[-+*]\s+\S/m,
|
|
420
|
+
/^ {0,3}>\s+\S/m,
|
|
421
|
+
/\[[^\]]+\]\([^)]+\)/,
|
|
422
|
+
/(?<![A-Za-z0-9])\*\*\S[\s\S]*?\S\*\*/,
|
|
423
|
+
/^ {0,3}\|.*\|\s*$/m,
|
|
424
|
+
];
|
|
425
|
+
return signals.filter((pattern) => pattern.test(sample)).length >= 2;
|
|
426
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Minimal WAV (RIFF) encoder for raw Float32 PCM.
|
|
3
|
+
*
|
|
4
|
+
* The browser path gets its WAV from `convertAudioBufferToWav`, which needs a Web
|
|
5
|
+
* Audio `AudioBuffer`. Node has no such thing: the Kokoro model hands back a bare
|
|
6
|
+
* `Float32Array`, and long documents are synthesized one chunk at a time and
|
|
7
|
+
* joined before encoding. Both of those are pure buffer work, so they live here
|
|
8
|
+
* with no platform dependency.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Number of bytes in the fixed RIFF/fmt/data header this encoder writes. */
|
|
12
|
+
const HEADER_BYTES = 44;
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Joins per-chunk sample buffers into one contiguous track.
|
|
16
|
+
*
|
|
17
|
+
* @param chunks Sample buffers in playback order.
|
|
18
|
+
* @param silenceSamples Samples of silence to insert between chunks, so the
|
|
19
|
+
* seams between separately synthesized chunks do not sound clipped.
|
|
20
|
+
*/
|
|
21
|
+
export function concatSamples(
|
|
22
|
+
chunks: Float32Array[],
|
|
23
|
+
silenceSamples = 0
|
|
24
|
+
): Float32Array {
|
|
25
|
+
const gaps = Math.max(chunks.length - 1, 0) * Math.max(silenceSamples, 0);
|
|
26
|
+
const total = chunks.reduce((sum, chunk) => sum + chunk.length, 0) + gaps;
|
|
27
|
+
const out = new Float32Array(total);
|
|
28
|
+
|
|
29
|
+
let offset = 0;
|
|
30
|
+
chunks.forEach((chunk, i) => {
|
|
31
|
+
out.set(chunk, offset);
|
|
32
|
+
offset += chunk.length;
|
|
33
|
+
if (i < chunks.length - 1) offset += Math.max(silenceSamples, 0);
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
return out;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Encodes Float32 samples (nominally -1..1) as a 16-bit PCM WAV file.
|
|
41
|
+
*
|
|
42
|
+
* @param samples Interleaved samples when `channels` is greater than 1.
|
|
43
|
+
* @param sampleRate Sample rate of the audio, e.g. 24000 for Kokoro.
|
|
44
|
+
* @param channels Channel count. Kokoro is mono.
|
|
45
|
+
*/
|
|
46
|
+
export function encodeWav(
|
|
47
|
+
samples: Float32Array,
|
|
48
|
+
sampleRate = 24000,
|
|
49
|
+
channels = 1
|
|
50
|
+
): ArrayBuffer {
|
|
51
|
+
const buffer = new ArrayBuffer(HEADER_BYTES + samples.length * 2);
|
|
52
|
+
const view = new DataView(buffer);
|
|
53
|
+
|
|
54
|
+
const writeString = (offset: number, value: string): void => {
|
|
55
|
+
for (let i = 0; i < value.length; i += 1) {
|
|
56
|
+
view.setUint8(offset + i, value.charCodeAt(i));
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
const byteRate = sampleRate * channels * 2;
|
|
61
|
+
|
|
62
|
+
writeString(0, "RIFF");
|
|
63
|
+
view.setUint32(4, 36 + samples.length * 2, true);
|
|
64
|
+
writeString(8, "WAVE");
|
|
65
|
+
writeString(12, "fmt ");
|
|
66
|
+
view.setUint32(16, 16, true); // PCM header size
|
|
67
|
+
view.setUint16(20, 1, true); // format: PCM
|
|
68
|
+
view.setUint16(22, channels, true);
|
|
69
|
+
view.setUint32(24, sampleRate, true);
|
|
70
|
+
view.setUint32(28, byteRate, true);
|
|
71
|
+
view.setUint16(32, channels * 2, true); // block align
|
|
72
|
+
view.setUint16(34, 16, true); // bits per sample
|
|
73
|
+
writeString(36, "data");
|
|
74
|
+
view.setUint32(40, samples.length * 2, true);
|
|
75
|
+
|
|
76
|
+
for (let i = 0; i < samples.length; i += 1) {
|
|
77
|
+
// Clamp before scaling: the model occasionally overshoots 1.0 and wrapping
|
|
78
|
+
// a 16-bit sample turns that into a loud click.
|
|
79
|
+
const sample = Math.max(-1, Math.min(1, samples[i]));
|
|
80
|
+
view.setInt16(HEADER_BYTES + i * 2, sample < 0 ? sample * 0x8000 : sample * 0x7fff, true);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
return buffer;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Duration of a Float32 track in seconds, for progress and log lines. */
|
|
87
|
+
export function wavDurationSeconds(samples: Float32Array, sampleRate: number): number {
|
|
88
|
+
return sampleRate > 0 ? samples.length / sampleRate : 0;
|
|
89
|
+
}
|