use-voice-control 0.1.85 → 0.1.87

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +206 -21
  2. package/bin/use-voice-control.mjs +20 -0
  3. package/dist/cli.js +374 -0
  4. package/dist/cli.js.map +1 -0
  5. package/dist/client/index.d.ts +1 -0
  6. package/dist/client/read-aloud.d.ts +14 -0
  7. package/dist/client.js +132 -157
  8. package/dist/client.js.map +1 -1
  9. package/dist/core/kokoro-node.d.ts +51 -0
  10. package/dist/core/kokoro.d.ts +6 -2
  11. package/dist/index.d.ts +2 -0
  12. package/dist/index.js +33 -76
  13. package/dist/index.js.map +1 -1
  14. package/dist/kokoro-node-DJ_Rxp_N.js +144 -0
  15. package/dist/kokoro-node-DJ_Rxp_N.js.map +1 -0
  16. package/dist/markdown.js +213 -0
  17. package/dist/markdown.js.map +1 -0
  18. package/dist/node/cli.d.ts +52 -0
  19. package/dist/node/document.d.ts +45 -0
  20. package/dist/node/index.d.ts +18 -0
  21. package/dist/node/render.d.ts +30 -0
  22. package/dist/node.js +27 -0
  23. package/dist/node.js.map +1 -0
  24. package/dist/react.js +13 -11
  25. package/dist/react.js.map +1 -1
  26. package/dist/semantic-split-CXhk-k1F.js +44 -0
  27. package/dist/semantic-split-CXhk-k1F.js.map +1 -0
  28. package/dist/types/types.d.ts +14 -1
  29. package/dist/utils/markdown-to-speech.d.ts +76 -0
  30. package/dist/utils/wav.d.ts +27 -0
  31. package/package.json +17 -4
  32. package/speech/client/index.ts +10 -0
  33. package/speech/client/read-aloud.ts +27 -1
  34. package/speech/core/kokoro-node.ts +164 -0
  35. package/speech/core/kokoro.ts +13 -66
  36. package/speech/index.ts +14 -0
  37. package/speech/node/cli.ts +509 -0
  38. package/speech/node/document.ts +126 -0
  39. package/speech/node/index.ts +64 -0
  40. package/speech/node/render.ts +79 -0
  41. package/speech/react/useReadAloud.ts +2 -0
  42. package/speech/types/types.ts +34 -5
  43. package/speech/utils/markdown-to-speech.ts +426 -0
  44. package/speech/utils/wav.ts +89 -0
@@ -0,0 +1 @@
1
+ {"version":3,"file":"cli.js","sources":["../speech/node/document.ts","../speech/node/render.ts","../speech/node/cli.ts"],"sourcesContent":["/**\n * @fileoverview Turning a document on disk (or on stdin) into speakable text.\n *\n * A `.md` file is converted with `markdownToSpeech` so the marks become\n * structure rather than words; a `.txt` file is passed through as it is. When\n * the source has no filename — piped stdin, a `--text` string — the format is\n * guessed from the content.\n */\nimport {\n looksLikeMarkdown,\n markdownToSpeech,\n type MarkdownToSpeechOptions,\n} from \"../utils/markdown-to-speech\";\n\nexport type DocumentFormat = \"markdown\" | \"text\";\n/** What the caller asked for; `auto` defers to the extension, then the content. */\nexport type RequestedFormat = DocumentFormat | \"auto\";\n\n/** Extensions read as Markdown. Everything else is treated as plain text. */\nexport const MARKDOWN_EXTENSIONS = [\".md\", \".markdown\", \".mdown\", \".mkd\", \".mdx\"];\n\n/** Extensions the CLI accepts without complaint. Others still work, with a warning. */\nexport const TEXT_EXTENSIONS = [\".txt\", \".text\", \"\"];\n\n/** Lowercased extension of a path, including the dot (`\"\"` when there is none). */\nexport function extensionOf(filename: string): string {\n const base = filename.split(/[\\\\/]/).pop() ?? \"\";\n const dot = base.lastIndexOf(\".\");\n return dot > 0 ? base.slice(dot).toLowerCase() : \"\";\n}\n\n/**\n * Decides how to read a document.\n *\n * An explicit `requested` format always wins — a `.txt` file full of Markdown is\n * still the user's call. Otherwise the extension decides, and content sniffing\n * is the last resort for input that arrived without a name.\n */\nexport function detectFormat(\n content: string,\n filename?: string,\n requested: RequestedFormat = \"auto\"\n): DocumentFormat {\n if (requested !== \"auto\") return requested;\n\n if (filename) {\n const extension = extensionOf(filename);\n if (MARKDOWN_EXTENSIONS.includes(extension)) return \"markdown\";\n if (TEXT_EXTENSIONS.includes(extension)) return \"text\";\n }\n\n return looksLikeMarkdown(content) ? \"markdown\" : \"text\";\n}\n\n/** Converts document content to the text that should be spoken. */\nexport function toSpeechText(\n content: string,\n format: DocumentFormat,\n options: MarkdownToSpeechOptions = {}\n): string {\n if (format === \"text\") {\n // Plain text is already speakable; only normalise line endings and the\n // trailing whitespace that would otherwise become an empty final chunk.\n return content.replace(/\\r\\n?/g, \"\\n\").trim();\n }\n return markdownToSpeech(content, options);\n}\n\nexport interface LoadedDocument {\n /** The speakable text, Markdown already converted. */\n text: string;\n /** How the content was read. */\n format: DocumentFormat;\n /** Where it came from, for log lines: a path, `stdin`, or `--text`. */\n source: string;\n}\n\nexport interface LoadDocumentOptions {\n /** Path to read. Use `-` for stdin. Mutually exclusive with `text`. */\n file?: string;\n /** Literal content to speak, instead of reading a file. */\n text?: string;\n format?: RequestedFormat;\n markdown?: MarkdownToSpeechOptions;\n /** Reads stdin. Injected so the CLI tests do not need a real pipe. */\n readStdin?: () => Promise<string>;\n}\n\n/**\n * Reads a document from a path, stdin, or an inline string and returns the text\n * to synthesize.\n */\nexport async function loadDocument(\n options: LoadDocumentOptions\n): Promise<LoadedDocument> {\n const { file, text, format = \"auto\", markdown = {} } = options;\n\n if (text !== undefined) {\n const resolved = detectFormat(text, undefined, format);\n return { text: toSpeechText(text, resolved, markdown), format: resolved, source: \"--text\" };\n }\n\n if (!file) throw new Error(\"No input: pass a file path, `-` for stdin, or --text\");\n\n if (file === \"-\") {\n const stdin = options.readStdin\n ? await options.readStdin()\n : await readStdinToString();\n const resolved = detectFormat(stdin, undefined, format);\n return { text: toSpeechText(stdin, resolved, markdown), format: resolved, source: \"stdin\" };\n }\n\n const { readFile } = await import(\"node:fs/promises\");\n const content = await readFile(file, \"utf8\");\n const resolved = detectFormat(content, file, format);\n return { text: toSpeechText(content, resolved, markdown), format: resolved, source: file };\n}\n\n/** Collects all of stdin as UTF-8 text. */\nexport async function readStdinToString(): Promise<string> {\n const chunks: Buffer[] = [];\n for await (const chunk of process.stdin) {\n chunks.push(Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk));\n }\n return Buffer.concat(chunks).toString(\"utf8\");\n}\n","/**\n * @fileoverview Rendering a loaded document to an audio file on disk.\n *\n * Sits between `document.ts` (what to say) and `core/kokoro-node.ts` (how to say\n * it): resolves the output path, runs the synthesizer, and writes the WAV.\n */\nimport {\n synthesizeSamples,\n type KokoroAudio,\n type KokoroNodeOptions,\n} from \"../core/kokoro-node\";\nimport { encodeWav, wavDurationSeconds } from \"../utils/wav\";\nimport { extensionOf, loadDocument, type LoadDocumentOptions } from \"./document\";\n\nexport interface RenderOptions extends LoadDocumentOptions, KokoroNodeOptions {\n /** Where to write the audio. `-` writes the WAV to stdout. */\n output?: string;\n /**\n * Replace the synthesizer. Used by the tests, and by hosts that already have a\n * model loaded and do not want a second copy.\n */\n synthesize?: (text: string, options: KokoroNodeOptions) => Promise<KokoroAudio>;\n}\n\nexport interface RenderResult {\n /** Path written, or `-` when the audio went to stdout. */\n output: string;\n /** The text that was spoken, after Markdown conversion. */\n text: string;\n format: \"markdown\" | \"text\";\n source: string;\n durationSeconds: number;\n bytes: number;\n}\n\n/** `notes.md` becomes `notes.wav`, next to the original. */\nexport function defaultOutputPath(input: string): string {\n const extension = extensionOf(input);\n const base = extension ? input.slice(0, -extension.length) : input;\n return `${base}.wav`;\n}\n\n/**\n * Reads a document, synthesizes it, and writes a WAV file.\n *\n * Returns the spoken text alongside the file details so a caller can show what\n * was read without converting the document a second time.\n */\nexport async function renderDocument(options: RenderOptions): Promise<RenderResult> {\n const document = await loadDocument(options);\n\n if (!document.text.trim()) {\n throw new Error(`Nothing to speak: ${document.source} has no readable text`);\n }\n\n const synthesize = options.synthesize ?? synthesizeSamples;\n const { samples, sampleRate } = await synthesize(document.text, options);\n const audio = encodeWav(samples, sampleRate);\n\n const output =\n options.output ??\n (options.file && options.file !== \"-\" ? defaultOutputPath(options.file) : \"out.wav\");\n\n if (output === \"-\") {\n process.stdout.write(Buffer.from(audio));\n } else {\n const { writeFile } = await import(\"node:fs/promises\");\n await writeFile(output, Buffer.from(audio));\n }\n\n return {\n output,\n text: document.text,\n format: document.format,\n source: document.source,\n durationSeconds: wavDurationSeconds(samples, sampleRate),\n bytes: audio.byteLength,\n };\n}\n","/**\n * @fileoverview The `use-voice-control` command line tool.\n *\n * Reads a Markdown or text file and speaks it to a WAV file:\n *\n * ```sh\n * npx use-voice-control README.md -o readme.wav\n * ```\n *\n * Argument parsing is a pure function (`parseArgs`) and every effect the command\n * has — writing lines, reading stdin, synthesizing — is injectable, so the whole\n * command is testable without loading a 90 MB model.\n */\nimport { describeKokoroVoice, KOKORO_VOICES } from \"../types/types\";\nimport type { KokoroDevice, KokoroDtype } from \"../core/kokoro-node\";\nimport type { MarkdownToSpeechOptions } from \"../utils/markdown-to-speech\";\nimport { loadDocument, type RequestedFormat } from \"./document\";\nimport { renderDocument, type RenderOptions } from \"./render\";\n\nexport type CliCommand = \"speak\" | \"print\" | \"help\" | \"version\" | \"list-voices\";\n\nexport interface CliOptions {\n command: CliCommand;\n /** Input path, or `-` for stdin. Undefined when `text` is set. */\n input?: string;\n /** Literal text passed with `--text`. */\n text?: string;\n output?: string;\n format: RequestedFormat;\n voice: string;\n speed: number;\n model?: string;\n dtype?: KokoroDtype;\n device?: KokoroDevice;\n maxChunkLength?: number;\n gapMs?: number;\n markdown: MarkdownToSpeechOptions;\n quiet: boolean;\n /** Everything wrong with the command line. Non-empty means do not run. */\n errors: string[];\n}\n\nexport interface CliIO {\n /** Writes a line to stdout. */\n log?: (line: string) => void;\n /** Writes a line to stderr — progress and errors go here so `-o -` stays clean. */\n error?: (line: string) => void;\n /** Overrides synthesis, for tests. */\n synthesize?: RenderOptions[\"synthesize\"];\n /** Overrides reading stdin, for tests. */\n readStdin?: () => Promise<string>;\n /** Whether stdin is a terminal. When false and no input is given, stdin is read. */\n stdinIsTTY?: boolean;\n}\n\nconst FORMATS: Record<string, RequestedFormat> = {\n auto: \"auto\",\n markdown: \"markdown\",\n md: \"markdown\",\n text: \"text\",\n txt: \"text\",\n plain: \"text\",\n};\n\nconst DTYPES: KokoroDtype[] = [\"fp32\", \"fp16\", \"q8\", \"q4\", \"q4f16\"];\nconst DEVICES: KokoroDevice[] = [\"wasm\", \"webgpu\", \"cpu\"];\n\nexport const USAGE = `use-voice-control — read a Markdown or text file aloud into an audio file\n\nUsage\n npx use-voice-control <file.md|file.txt|-> [options]\n npx use-voice-control --text \"Hello there\" -o hello.wav\n cat notes.md | npx use-voice-control - -o notes.wav\n\nMarkdown is converted before it is spoken: \"#\", \"**\" and the rest are not read\nout, headings become their own spoken lines, links keep their text, and fenced\ncode blocks are announced instead of being spelled out.\n\nInput\n <file> File to read. Use \"-\" to read stdin.\n --text <string> Speak this string instead of reading a file.\n -f, --format <fmt> auto (default), markdown, or text.\n\nOutput\n -o, --out <file> Audio file to write. Default: the input path with a\n .wav extension. Use \"-\" to write the WAV to stdout.\n -p, --print Print the speakable text and exit — no model, no audio.\n Useful for checking the Markdown conversion.\n -q, --quiet No progress output.\n\nVoice\n -v, --voice <id> Voice id. Default af_heart. See --list-voices.\n -s, --speed <n> Speaking rate, 0.5-2. Default 1.\n --list-voices Print the available voices and exit.\n\nMarkdown handling\n --headings <mode> text (default) | announce | skip\n --code <mode> announce (default) | read | skip\n --links <mode> text (default) | text-and-url\n --tables <mode> rows (default) | skip\n --front-matter Read the YAML front matter instead of skipping it.\n\nModel\n --model <id> Hugging Face model id.\n Default onnx-community/Kokoro-82M-v1.0-ONNX.\n --dtype <type> fp32 | fp16 | q8 (default) | q4 | q4f16\n --device <device> cpu (default) | wasm | webgpu\n --chunk <chars> Target characters per synthesis chunk. Default 400.\n --gap <ms> Silence between chunks. Default 120.\n\nOther\n -h, --help Show this help.\n -V, --version Print the package version.\n\nThe first run downloads the Kokoro weights (about 90 MB at the default q8) into\nthe Hugging Face cache; later runs are offline. Speech is synthesized locally —\nno text leaves the machine.`;\n\n/** Reads the next value for a flag, recording an error when it is missing. */\nfunction takeValue(\n argv: string[],\n index: number,\n flag: string,\n errors: string[]\n): { value?: string; next: number } {\n const value = argv[index + 1];\n if (value === undefined || value.startsWith(\"-\")) {\n errors.push(`${flag} needs a value`);\n return { next: index + 1 };\n }\n return { value, next: index + 1 };\n}\n\n/** Parses a numeric flag, recording an error when it is not a finite number. */\nfunction toNumber(value: string | undefined, flag: string, errors: string[]): number | undefined {\n if (value === undefined) return undefined;\n const parsed = Number(value);\n if (!Number.isFinite(parsed)) {\n errors.push(`${flag} expects a number, got \"${value}\"`);\n return undefined;\n }\n return parsed;\n}\n\n/** Parses an enum flag, recording an error when the value is not in `allowed`. */\nfunction toChoice<T extends string>(\n value: string | undefined,\n allowed: readonly T[],\n flag: string,\n errors: string[]\n): T | undefined {\n if (value === undefined) return undefined;\n if (!allowed.includes(value as T)) {\n errors.push(`${flag} expects one of ${allowed.join(\", \")} — got \"${value}\"`);\n return undefined;\n }\n return value as T;\n}\n\n/**\n * Parses the command line. Pure: it reads nothing and writes nothing, so the\n * tests can assert on the whole option set.\n */\nexport function parseArgs(argv: string[]): CliOptions {\n const errors: string[] = [];\n const markdown: MarkdownToSpeechOptions = {};\n const options: CliOptions = {\n command: \"speak\",\n format: \"auto\",\n voice: \"af_heart\",\n speed: 1,\n markdown,\n quiet: false,\n errors,\n };\n\n let positionalsOnly = false;\n\n for (let i = 0; i < argv.length; i += 1) {\n const arg = argv[i];\n\n if (positionalsOnly || !arg.startsWith(\"-\") || arg === \"-\") {\n if (options.input !== undefined) {\n errors.push(`unexpected extra input \"${arg}\" — pass one file at a time`);\n } else {\n options.input = arg;\n }\n continue;\n }\n\n if (arg === \"--\") {\n positionalsOnly = true;\n continue;\n }\n\n // `--voice=af_bella` is the same as `--voice af_bella`.\n const equals = arg.indexOf(\"=\");\n if (equals > 1) {\n argv.splice(i, 1, arg.slice(0, equals), arg.slice(equals + 1));\n i -= 1;\n continue;\n }\n\n switch (arg) {\n case \"-h\":\n case \"--help\":\n options.command = \"help\";\n return options;\n\n case \"-V\":\n case \"--version\":\n options.command = \"version\";\n return options;\n\n case \"--list-voices\":\n options.command = \"list-voices\";\n return options;\n\n case \"-p\":\n case \"--print\":\n case \"--dry-run\":\n options.command = \"print\";\n break;\n\n case \"-q\":\n case \"--quiet\":\n options.quiet = true;\n break;\n\n case \"--front-matter\":\n markdown.frontMatter = true;\n break;\n\n case \"-o\":\n case \"--out\":\n case \"--output\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n options.output = value;\n i = next;\n break;\n }\n\n case \"--text\": {\n // Unlike every other value, this one may legitimately start with \"-\".\n const value = argv[i + 1];\n if (value === undefined) errors.push(\"--text needs a value\");\n else options.text = value;\n i += 1;\n break;\n }\n\n case \"-f\":\n case \"--format\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n if (value !== undefined) {\n const format = FORMATS[value.toLowerCase()];\n if (!format) {\n errors.push(`--format expects auto, markdown or text — got \"${value}\"`);\n } else {\n options.format = format;\n }\n }\n break;\n }\n\n case \"-v\":\n case \"--voice\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n if (value !== undefined) {\n if (!KOKORO_VOICES.includes(value as (typeof KOKORO_VOICES)[number])) {\n errors.push(`unknown voice \"${value}\" — run --list-voices to see them all`);\n } else {\n options.voice = value;\n }\n }\n break;\n }\n\n case \"-s\":\n case \"--speed\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n const speed = toNumber(value, arg, errors);\n if (speed !== undefined) {\n if (speed < 0.5 || speed > 2) errors.push(\"--speed must be between 0.5 and 2\");\n else options.speed = speed;\n }\n break;\n }\n\n case \"--model\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n options.model = value;\n i = next;\n break;\n }\n\n case \"--dtype\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n options.dtype = toChoice(value, DTYPES, arg, errors) ?? options.dtype;\n break;\n }\n\n case \"--device\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n options.device = toChoice(value, DEVICES, arg, errors) ?? options.device;\n break;\n }\n\n case \"--chunk\":\n case \"--chunk-length\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n const chunk = toNumber(value, arg, errors);\n if (chunk !== undefined) {\n if (chunk < 40) errors.push(\"--chunk must be at least 40 characters\");\n else options.maxChunkLength = chunk;\n }\n break;\n }\n\n case \"--gap\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n const gap = toNumber(value, arg, errors);\n if (gap !== undefined) {\n if (gap < 0) errors.push(\"--gap cannot be negative\");\n else options.gapMs = gap;\n }\n break;\n }\n\n case \"--headings\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n markdown.headings =\n toChoice(value, [\"text\", \"announce\", \"skip\"] as const, arg, errors) ??\n markdown.headings;\n break;\n }\n\n case \"--code\":\n case \"--code-blocks\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n markdown.codeBlocks =\n toChoice(value, [\"announce\", \"read\", \"skip\"] as const, arg, errors) ??\n markdown.codeBlocks;\n break;\n }\n\n case \"--links\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n markdown.links =\n toChoice(value, [\"text\", \"text-and-url\"] as const, arg, errors) ?? markdown.links;\n break;\n }\n\n case \"--tables\": {\n const { value, next } = takeValue(argv, i, arg, errors);\n i = next;\n markdown.tables =\n toChoice(value, [\"rows\", \"skip\"] as const, arg, errors) ?? markdown.tables;\n break;\n }\n\n default:\n errors.push(`unknown option \"${arg}\" — run --help to see the options`);\n }\n }\n\n if (options.text !== undefined && options.input !== undefined) {\n errors.push(\"pass either a file or --text, not both\");\n }\n\n return options;\n}\n\n/** The `--list-voices` table, derived from the ids so it cannot drift. */\nexport function formatVoiceList(): string {\n const rows = KOKORO_VOICES.map((id) => {\n const voice = describeKokoroVoice(id);\n return ` ${id.padEnd(14)}${voice.name.padEnd(12)}${voice.gender.padEnd(8)}${voice.accent}`;\n });\n return [`${KOKORO_VOICES.length} Kokoro voices:`, ...rows].join(\"\\n\");\n}\n\n/** Reads the package version, so `--version` matches what npx installed. */\nasync function readVersion(): Promise<string> {\n try {\n const { readFile } = await import(\"node:fs/promises\");\n // From `dist/cli.js` the manifest is one level up; from a source checkout it\n // is two. Try both before giving up.\n for (const relative of [\"../package.json\", \"../../package.json\"]) {\n try {\n const url = new URL(relative, import.meta.url);\n const manifest = JSON.parse(await readFile(url, \"utf8\"));\n if (manifest?.name === \"use-voice-control\" && manifest.version) return manifest.version;\n } catch {\n // Try the next candidate.\n }\n }\n } catch {\n // No filesystem (bundled for a runtime without fs) — fall through.\n }\n return \"unknown\";\n}\n\n/**\n * Runs the command line and resolves to the process exit code.\n *\n * @param argv Arguments after the executable and script, i.e. `process.argv.slice(2)`.\n * @param io Injection points for output, stdin and synthesis.\n */\nexport async function runCli(argv: string[], io: CliIO = {}): Promise<number> {\n const log = io.log ?? ((line: string) => process.stdout.write(`${line}\\n`));\n const warn = io.error ?? ((line: string) => process.stderr.write(`${line}\\n`));\n\n const options = parseArgs([...argv]);\n\n if (options.command === \"help\") {\n log(USAGE);\n return 0;\n }\n\n if (options.command === \"version\") {\n log(await readVersion());\n return 0;\n }\n\n if (options.command === \"list-voices\") {\n log(formatVoiceList());\n return 0;\n }\n\n if (options.errors.length > 0) {\n options.errors.forEach((message) => warn(`use-voice-control: ${message}`));\n warn(\"Run `npx use-voice-control --help` for usage.\");\n return 1;\n }\n\n // With nothing named on the command line, a pipe is the intended input.\n const stdinIsTTY = io.stdinIsTTY ?? Boolean(process.stdin.isTTY);\n const input =\n options.input ?? (options.text === undefined && !stdinIsTTY ? \"-\" : undefined);\n\n if (input === undefined && options.text === undefined) {\n warn(\"use-voice-control: no input — pass a file, pipe text in, or use --text\");\n warn(\"Run `npx use-voice-control --help` for usage.\");\n return 1;\n }\n\n const shared = {\n file: input,\n text: options.text,\n format: options.format,\n markdown: options.markdown,\n readStdin: io.readStdin,\n };\n\n try {\n if (options.command === \"print\") {\n const document = await loadDocument(shared);\n log(document.text);\n return 0;\n }\n\n const started = Date.now();\n const result = await renderDocument({\n ...shared,\n output: options.output,\n voice: options.voice,\n speed: options.speed,\n model: options.model,\n dtype: options.dtype,\n device: options.device,\n maxChunkLength: options.maxChunkLength,\n gapMs: options.gapMs,\n synthesize: io.synthesize,\n onModelProgress: options.quiet\n ? undefined\n : (progress: any) => {\n if (progress?.status === \"progress\" && progress.file && progress.total) {\n const done = Math.round(((progress.loaded ?? 0) / progress.total) * 100);\n warn(`downloading ${progress.file}: ${done}%`);\n }\n },\n onChunk: options.quiet\n ? undefined\n : ({ index, total }) => warn(`speaking chunk ${index + 1}/${total}`),\n });\n\n if (!options.quiet) {\n const seconds = result.durationSeconds.toFixed(1);\n const elapsed = ((Date.now() - started) / 1000).toFixed(1);\n const target = result.output === \"-\" ? \"stdout\" : result.output;\n warn(`wrote ${target} — ${seconds}s of audio from ${result.source} in ${elapsed}s`);\n }\n return 0;\n } catch (error) {\n warn(`use-voice-control: ${error instanceof Error ? error.message : String(error)}`);\n return 1;\n }\n}\n"],"names":["MARKDOWN_EXTENSIONS","TEXT_EXTENSIONS","extensionOf","filename","base","dot","detectFormat","content","requested","extension","looksLikeMarkdown","toSpeechText","format","options","markdownToSpeech","loadDocument","file","text","markdown","resolved","stdin","readStdinToString","readFile","chunks","chunk","defaultOutputPath","input","renderDocument","document","synthesize","synthesizeSamples","samples","sampleRate","audio","encodeWav","output","writeFile","wavDurationSeconds","FORMATS","DTYPES","DEVICES","USAGE","takeValue","argv","index","flag","errors","value","toNumber","parsed","toChoice","allowed","parseArgs","positionalsOnly","arg","equals","next","KOKORO_VOICES","speed","gap","formatVoiceList","rows","id","voice","describeKokoroVoice","readVersion","relative","url","manifest","runCli","io","log","line","warn","message","stdinIsTTY","shared","started","result","progress","done","total","seconds","elapsed","target","error"],"mappings":";;AAmBO,MAAMA,IAAsB,CAAC,OAAO,aAAa,UAAU,QAAQ,MAAM,GAGnEC,IAAkB,CAAC,QAAQ,SAAS,EAAE;AAG5C,SAASC,EAAYC,GAA0B;AACpD,QAAMC,IAAOD,EAAS,MAAM,OAAO,EAAE,SAAS,IACxCE,IAAMD,EAAK,YAAY,GAAG;AAChC,SAAOC,IAAM,IAAID,EAAK,MAAMC,CAAG,EAAE,gBAAgB;AACnD;AASO,SAASC,EACdC,GACAJ,GACAK,IAA6B,QACb;AAChB,MAAIA,MAAc,OAAQ,QAAOA;AAEjC,MAAIL,GAAU;AACZ,UAAMM,IAAYP,EAAYC,CAAQ;AACtC,QAAIH,EAAoB,SAASS,CAAS,EAAG,QAAO;AACpD,QAAIR,EAAgB,SAASQ,CAAS,EAAG,QAAO;AAAA,EAClD;AAEA,SAAOC,EAAkBH,CAAO,IAAI,aAAa;AACnD;AAGO,SAASI,EACdJ,GACAK,GACAC,IAAmC,CAAA,GAC3B;AACR,SAAID,MAAW,SAGNL,EAAQ,QAAQ,UAAU;AAAA,CAAI,EAAE,KAAA,IAElCO,EAAiBP,GAASM,CAAO;AAC1C;AA0BA,eAAsBE,EACpBF,GACyB;AACzB,QAAM,EAAE,MAAAG,GAAM,MAAAC,GAAM,QAAAL,IAAS,QAAQ,UAAAM,IAAW,CAAA,MAAOL;AAEvD,MAAII,MAAS,QAAW;AACtB,UAAME,IAAWb,EAAaW,GAAM,QAAWL,CAAM;AACrD,WAAO,EAAE,MAAMD,EAAaM,GAAME,GAAUD,CAAQ,GAAG,QAAQC,GAAU,QAAQ,SAAA;AAAA,EACnF;AAEA,MAAI,CAACH,EAAM,OAAM,IAAI,MAAM,sDAAsD;AAEjF,MAAIA,MAAS,KAAK;AAChB,UAAMI,IAAQP,EAAQ,YAClB,MAAMA,EAAQ,UAAA,IACd,MAAMQ,EAAA,GACJF,IAAWb,EAAac,GAAO,QAAWR,CAAM;AACtD,WAAO,EAAE,MAAMD,EAAaS,GAAOD,GAAUD,CAAQ,GAAG,QAAQC,GAAU,QAAQ,QAAA;AAAA,EACpF;AAEA,QAAM,EAAE,UAAAG,EAAA,IAAa,MAAM,OAAO,kBAAkB,GAC9Cf,IAAU,MAAMe,EAASN,GAAM,MAAM,GACrCG,IAAWb,EAAaC,GAASS,GAAMJ,CAAM;AACnD,SAAO,EAAE,MAAMD,EAAaJ,GAASY,GAAUD,CAAQ,GAAG,QAAQC,GAAU,QAAQH,EAAA;AACtF;AAGA,eAAsBK,IAAqC;AACzD,QAAME,IAAmB,CAAA;AACzB,mBAAiBC,KAAS,QAAQ;AAChC,IAAAD,EAAO,KAAK,OAAO,SAASC,CAAK,IAAIA,IAAQ,OAAO,KAAKA,CAAK,CAAC;AAEjE,SAAO,OAAO,OAAOD,CAAM,EAAE,SAAS,MAAM;AAC9C;ACzFO,SAASE,EAAkBC,GAAuB;AACvD,QAAMjB,IAAYP,EAAYwB,CAAK;AAEnC,SAAO,GADMjB,IAAYiB,EAAM,MAAM,GAAG,CAACjB,EAAU,MAAM,IAAIiB,CAC/C;AAChB;AAQA,eAAsBC,EAAed,GAA+C;AAClF,QAAMe,IAAW,MAAMb,EAAaF,CAAO;AAE3C,MAAI,CAACe,EAAS,KAAK;AACjB,UAAM,IAAI,MAAM,qBAAqBA,EAAS,MAAM,uBAAuB;AAG7E,QAAMC,IAAahB,EAAQ,cAAciB,GACnC,EAAE,SAAAC,GAAS,YAAAC,EAAA,IAAe,MAAMH,EAAWD,EAAS,MAAMf,CAAO,GACjEoB,IAAQC,EAAUH,GAASC,CAAU,GAErCG,IACJtB,EAAQ,WACPA,EAAQ,QAAQA,EAAQ,SAAS,MAAMY,EAAkBZ,EAAQ,IAAI,IAAI;AAE5E,MAAIsB,MAAW;AACb,YAAQ,OAAO,MAAM,OAAO,KAAKF,CAAK,CAAC;AAAA,OAClC;AACL,UAAM,EAAE,WAAAG,EAAA,IAAc,MAAM,OAAO,kBAAkB;AACrD,UAAMA,EAAUD,GAAQ,OAAO,KAAKF,CAAK,CAAC;AAAA,EAC5C;AAEA,SAAO;AAAA,IACL,QAAAE;AAAA,IACA,MAAMP,EAAS;AAAA,IACf,QAAQA,EAAS;AAAA,IACjB,QAAQA,EAAS;AAAA,IACjB,iBAAiBS,EAAmBN,GAASC,CAAU;AAAA,IACvD,OAAOC,EAAM;AAAA,EAAA;AAEjB;ACvBA,MAAMK,IAA2C;AAAA,EAC/C,MAAM;AAAA,EACN,UAAU;AAAA,EACV,IAAI;AAAA,EACJ,MAAM;AAAA,EACN,KAAK;AAAA,EACL,OAAO;AACT,GAEMC,IAAwB,CAAC,QAAQ,QAAQ,MAAM,MAAM,OAAO,GAC5DC,IAA0B,CAAC,QAAQ,UAAU,KAAK,GAE3CC,IAAQ;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAoDrB,SAASC,EACPC,GACAC,GACAC,GACAC,GACkC;AAClC,QAAMC,IAAQJ,EAAKC,IAAQ,CAAC;AAC5B,SAAIG,MAAU,UAAaA,EAAM,WAAW,GAAG,KAC7CD,EAAO,KAAK,GAAGD,CAAI,gBAAgB,GAC5B,EAAE,MAAMD,IAAQ,EAAA,KAElB,EAAE,OAAAG,GAAO,MAAMH,IAAQ,EAAA;AAChC;AAGA,SAASI,EAASD,GAA2BF,GAAcC,GAAsC;AAC/F,MAAIC,MAAU,OAAW;AACzB,QAAME,IAAS,OAAOF,CAAK;AAC3B,MAAI,CAAC,OAAO,SAASE,CAAM,GAAG;AAC5B,IAAAH,EAAO,KAAK,GAAGD,CAAI,2BAA2BE,CAAK,GAAG;AACtD;AAAA,EACF;AACA,SAAOE;AACT;AAGA,SAASC,EACPH,GACAI,GACAN,GACAC,GACe;AACf,MAAIC,MAAU,QACd;AAAA,QAAI,CAACI,EAAQ,SAASJ,CAAU,GAAG;AACjC,MAAAD,EAAO,KAAK,GAAGD,CAAI,mBAAmBM,EAAQ,KAAK,IAAI,CAAC,WAAWJ,CAAK,GAAG;AAC3E;AAAA,IACF;AACA,WAAOA;AAAA;AACT;AAMO,SAASK,EAAUT,GAA4B;AACpD,QAAMG,IAAmB,CAAA,GACnB5B,IAAoC,CAAA,GACpCL,IAAsB;AAAA,IAC1B,SAAS;AAAA,IACT,QAAQ;AAAA,IACR,OAAO;AAAA,IACP,OAAO;AAAA,IACP,UAAAK;AAAA,IACA,OAAO;AAAA,IACP,QAAA4B;AAAA,EAAA;AAGF,MAAIO,IAAkB;AAEtB,WAAS,IAAI,GAAG,IAAIV,EAAK,QAAQ,KAAK,GAAG;AACvC,UAAMW,IAAMX,EAAK,CAAC;AAElB,QAAIU,KAAmB,CAACC,EAAI,WAAW,GAAG,KAAKA,MAAQ,KAAK;AAC1D,MAAIzC,EAAQ,UAAU,SACpBiC,EAAO,KAAK,2BAA2BQ,CAAG,6BAA6B,IAEvEzC,EAAQ,QAAQyC;AAElB;AAAA,IACF;AAEA,QAAIA,MAAQ,MAAM;AAChB,MAAAD,IAAkB;AAClB;AAAA,IACF;AAGA,UAAME,IAASD,EAAI,QAAQ,GAAG;AAC9B,QAAIC,IAAS,GAAG;AACd,MAAAZ,EAAK,OAAO,GAAG,GAAGW,EAAI,MAAM,GAAGC,CAAM,GAAGD,EAAI,MAAMC,IAAS,CAAC,CAAC,GAC7D,KAAK;AACL;AAAA,IACF;AAEA,YAAQD,GAAA;AAAA,MACN,KAAK;AAAA,MACL,KAAK;AACH,eAAAzC,EAAQ,UAAU,QACXA;AAAA,MAET,KAAK;AAAA,MACL,KAAK;AACH,eAAAA,EAAQ,UAAU,WACXA;AAAA,MAET,KAAK;AACH,eAAAA,EAAQ,UAAU,eACXA;AAAA,MAET,KAAK;AAAA,MACL,KAAK;AAAA,MACL,KAAK;AACH,QAAAA,EAAQ,UAAU;AAClB;AAAA,MAEF,KAAK;AAAA,MACL,KAAK;AACH,QAAAA,EAAQ,QAAQ;AAChB;AAAA,MAEF,KAAK;AACH,QAAAK,EAAS,cAAc;AACvB;AAAA,MAEF,KAAK;AAAA,MACL,KAAK;AAAA,MACL,KAAK,YAAY;AACf,cAAM,EAAE,OAAA6B,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,QAAAjC,EAAQ,SAASkC,GACjB,IAAIS;AACJ;AAAA,MACF;AAAA,MAEA,KAAK,UAAU;AAEb,cAAMT,IAAQJ,EAAK,IAAI,CAAC;AACxB,QAAII,MAAU,SAAWD,EAAO,KAAK,sBAAsB,MAC9C,OAAOC,GACpB,KAAK;AACL;AAAA,MACF;AAAA,MAEA,KAAK;AAAA,MACL,KAAK,YAAY;AACf,cAAM,EAAE,OAAAA,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AAEtD,YADA,IAAIU,GACAT,MAAU,QAAW;AACvB,gBAAMnC,IAAS0B,EAAQS,EAAM,YAAA,CAAa;AAC1C,UAAKnC,IAGHC,EAAQ,SAASD,IAFjBkC,EAAO,KAAK,kDAAkDC,CAAK,GAAG;AAAA,QAI1E;AACA;AAAA,MACF;AAAA,MAEA,KAAK;AAAA,MACL,KAAK,WAAW;AACd,cAAM,EAAE,OAAAA,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACAT,MAAU,WACPU,EAAc,SAASV,CAAuC,IAGjElC,EAAQ,QAAQkC,IAFhBD,EAAO,KAAK,kBAAkBC,CAAK,uCAAuC;AAK9E;AAAA,MACF;AAAA,MAEA,KAAK;AAAA,MACL,KAAK,WAAW;AACd,cAAM,EAAE,OAAAA,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU;AACJ,cAAME,IAAQV,EAASD,GAAOO,GAAKR,CAAM;AACzC,QAAIY,MAAU,WACRA,IAAQ,OAAOA,IAAQ,IAAGZ,EAAO,KAAK,mCAAmC,MAChE,QAAQY;AAEvB;AAAA,MACF;AAAA,MAEA,KAAK,WAAW;AACd,cAAM,EAAE,OAAAX,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,QAAAjC,EAAQ,QAAQkC,GAChB,IAAIS;AACJ;AAAA,MACF;AAAA,MAEA,KAAK,WAAW;AACd,cAAM,EAAE,OAAAT,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJ3C,EAAQ,QAAQqC,EAASH,GAAOR,GAAQe,GAAKR,CAAM,KAAKjC,EAAQ;AAChE;AAAA,MACF;AAAA,MAEA,KAAK,YAAY;AACf,cAAM,EAAE,OAAAkC,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJ3C,EAAQ,SAASqC,EAASH,GAAOP,GAASc,GAAKR,CAAM,KAAKjC,EAAQ;AAClE;AAAA,MACF;AAAA,MAEA,KAAK;AAAA,MACL,KAAK,kBAAkB;AACrB,cAAM,EAAE,OAAAkC,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU;AACJ,cAAMhC,IAAQwB,EAASD,GAAOO,GAAKR,CAAM;AACzC,QAAItB,MAAU,WACRA,IAAQ,KAAIsB,EAAO,KAAK,wCAAwC,MACvD,iBAAiBtB;AAEhC;AAAA,MACF;AAAA,MAEA,KAAK,SAAS;AACZ,cAAM,EAAE,OAAAuB,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU;AACJ,cAAMG,IAAMX,EAASD,GAAOO,GAAKR,CAAM;AACvC,QAAIa,MAAQ,WACNA,IAAM,IAAGb,EAAO,KAAK,0BAA0B,MACtC,QAAQa;AAEvB;AAAA,MACF;AAAA,MAEA,KAAK,cAAc;AACjB,cAAM,EAAE,OAAAZ,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJtC,EAAS,WACPgC,EAASH,GAAO,CAAC,QAAQ,YAAY,MAAM,GAAYO,GAAKR,CAAM,KAClE5B,EAAS;AACX;AAAA,MACF;AAAA,MAEA,KAAK;AAAA,MACL,KAAK,iBAAiB;AACpB,cAAM,EAAE,OAAA6B,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJtC,EAAS,aACPgC,EAASH,GAAO,CAAC,YAAY,QAAQ,MAAM,GAAYO,GAAKR,CAAM,KAClE5B,EAAS;AACX;AAAA,MACF;AAAA,MAEA,KAAK,WAAW;AACd,cAAM,EAAE,OAAA6B,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJtC,EAAS,QACPgC,EAASH,GAAO,CAAC,QAAQ,cAAc,GAAYO,GAAKR,CAAM,KAAK5B,EAAS;AAC9E;AAAA,MACF;AAAA,MAEA,KAAK,YAAY;AACf,cAAM,EAAE,OAAA6B,GAAO,MAAAS,MAASd,EAAUC,GAAM,GAAGW,GAAKR,CAAM;AACtD,YAAIU,GACJtC,EAAS,SACPgC,EAASH,GAAO,CAAC,QAAQ,MAAM,GAAYO,GAAKR,CAAM,KAAK5B,EAAS;AACtE;AAAA,MACF;AAAA,MAEA;AACE,QAAA4B,EAAO,KAAK,mBAAmBQ,CAAG,mCAAmC;AAAA,IAAA;AAAA,EAE3E;AAEA,SAAIzC,EAAQ,SAAS,UAAaA,EAAQ,UAAU,UAClDiC,EAAO,KAAK,wCAAwC,GAG/CjC;AACT;AAGO,SAAS+C,IAA0B;AACxC,QAAMC,IAAOJ,EAAc,IAAI,CAACK,MAAO;AACrC,UAAMC,IAAQC,EAAoBF,CAAE;AACpC,WAAO,KAAKA,EAAG,OAAO,EAAE,CAAC,GAAGC,EAAM,KAAK,OAAO,EAAE,CAAC,GAAGA,EAAM,OAAO,OAAO,CAAC,CAAC,GAAGA,EAAM,MAAM;AAAA,EAC3F,CAAC;AACD,SAAO,CAAC,GAAGN,EAAc,MAAM,mBAAmB,GAAGI,CAAI,EAAE,KAAK;AAAA,CAAI;AACtE;AAGA,eAAeI,IAA+B;AAC5C,MAAI;AACF,UAAM,EAAE,UAAA3C,EAAA,IAAa,MAAM,OAAO,kBAAkB;AAGpD,eAAW4C,KAAY,CAAC,mBAAmB,oBAAoB;AAC7D,UAAI;AACF,cAAMC,IAAM,IAAI,IAAID,GAAU,YAAY,GAAG,GACvCE,IAAW,KAAK,MAAM,MAAM9C,EAAS6C,GAAK,MAAM,CAAC;AACvD,aAAIC,KAAA,gBAAAA,EAAU,UAAS,uBAAuBA,EAAS,gBAAgBA,EAAS;AAAA,MAClF,QAAQ;AAAA,MAER;AAAA,EAEJ,QAAQ;AAAA,EAER;AACA,SAAO;AACT;AAQA,eAAsBC,EAAO1B,GAAgB2B,IAAY,IAAqB;AAC5E,QAAMC,IAAMD,EAAG,QAAQ,CAACE,MAAiB,QAAQ,OAAO,MAAM,GAAGA,CAAI;AAAA,CAAI,IACnEC,IAAOH,EAAG,UAAU,CAACE,MAAiB,QAAQ,OAAO,MAAM,GAAGA,CAAI;AAAA,CAAI,IAEtE3D,IAAUuC,EAAU,CAAC,GAAGT,CAAI,CAAC;AAEnC,MAAI9B,EAAQ,YAAY;AACtB,WAAA0D,EAAI9B,CAAK,GACF;AAGT,MAAI5B,EAAQ,YAAY;AACtB,WAAA0D,EAAI,MAAMN,GAAa,GAChB;AAGT,MAAIpD,EAAQ,YAAY;AACtB,WAAA0D,EAAIX,GAAiB,GACd;AAGT,MAAI/C,EAAQ,OAAO,SAAS;AAC1B,WAAAA,EAAQ,OAAO,QAAQ,CAAC6D,MAAYD,EAAK,sBAAsBC,CAAO,EAAE,CAAC,GACzED,EAAK,+CAA+C,GAC7C;AAIT,QAAME,IAAaL,EAAG,cAAc,EAAQ,QAAQ,MAAM,OACpD5C,IACJb,EAAQ,UAAUA,EAAQ,SAAS,UAAa,CAAC8D,IAAa,MAAM;AAEtE,MAAIjD,MAAU,UAAab,EAAQ,SAAS;AAC1C,WAAA4D,EAAK,wEAAwE,GAC7EA,EAAK,+CAA+C,GAC7C;AAGT,QAAMG,IAAS;AAAA,IACb,MAAMlD;AAAA,IACN,MAAMb,EAAQ;AAAA,IACd,QAAQA,EAAQ;AAAA,IAChB,UAAUA,EAAQ;AAAA,IAClB,WAAWyD,EAAG;AAAA,EAAA;AAGhB,MAAI;AACF,QAAIzD,EAAQ,YAAY,SAAS;AAC/B,YAAMe,IAAW,MAAMb,EAAa6D,CAAM;AAC1C,aAAAL,EAAI3C,EAAS,IAAI,GACV;AAAA,IACT;AAEA,UAAMiD,IAAU,KAAK,IAAA,GACfC,IAAS,MAAMnD,EAAe;AAAA,MAClC,GAAGiD;AAAA,MACH,QAAQ/D,EAAQ;AAAA,MAChB,OAAOA,EAAQ;AAAA,MACf,OAAOA,EAAQ;AAAA,MACf,OAAOA,EAAQ;AAAA,MACf,OAAOA,EAAQ;AAAA,MACf,QAAQA,EAAQ;AAAA,MAChB,gBAAgBA,EAAQ;AAAA,MACxB,OAAOA,EAAQ;AAAA,MACf,YAAYyD,EAAG;AAAA,MACf,iBAAiBzD,EAAQ,QACrB,SACA,CAACkE,MAAkB;AACjB,aAAIA,KAAA,gBAAAA,EAAU,YAAW,cAAcA,EAAS,QAAQA,EAAS,OAAO;AACtE,gBAAMC,IAAO,KAAK,OAAQD,EAAS,UAAU,KAAKA,EAAS,QAAS,GAAG;AACvE,UAAAN,EAAK,eAAeM,EAAS,IAAI,KAAKC,CAAI,GAAG;AAAA,QAC/C;AAAA,MACF;AAAA,MACJ,SAASnE,EAAQ,QACb,SACA,CAAC,EAAE,OAAA+B,GAAO,OAAAqC,EAAA,MAAYR,EAAK,kBAAkB7B,IAAQ,CAAC,IAAIqC,CAAK,EAAE;AAAA,IAAA,CACtE;AAED,QAAI,CAACpE,EAAQ,OAAO;AAClB,YAAMqE,IAAUJ,EAAO,gBAAgB,QAAQ,CAAC,GAC1CK,MAAY,KAAK,IAAA,IAAQN,KAAW,KAAM,QAAQ,CAAC,GACnDO,IAASN,EAAO,WAAW,MAAM,WAAWA,EAAO;AACzD,MAAAL,EAAK,SAASW,CAAM,MAAMF,CAAO,mBAAmBJ,EAAO,MAAM,OAAOK,CAAO,GAAG;AAAA,IACpF;AACA,WAAO;AAAA,EACT,SAASE,GAAO;AACd,WAAAZ,EAAK,sBAAsBY,aAAiB,QAAQA,EAAM,UAAU,OAAOA,CAAK,CAAC,EAAE,GAC5E;AAAA,EACT;AACF;"}
@@ -5,4 +5,5 @@
5
5
  */
6
6
  export { ReadAloudController, type ReadAloudChunk, type ReadAloudOptions, type ReadAloudState, type SynthesizeFn, } from './read-aloud';
7
7
  export { LiveTranscriber, isTranscriptionSupported, type LiveTranscriberOptions, type TranscriberEngine, } from './live-transcriber';
8
+ export { looksLikeMarkdown, markdownToSpeech, markdownToSpeechSegments, stripInlineMarkdown, type MarkdownToSpeechOptions, type SpeechSegment, type SpeechSegmentType, } from '../utils/markdown-to-speech';
8
9
  export type { TTSProvider, KokoroVoice, DeepgramSpeaker } from '../types/types';
@@ -1,4 +1,5 @@
1
1
  import { TTSProvider } from '../types/types';
2
+ import { MarkdownToSpeechOptions } from '../utils/markdown-to-speech';
2
3
  export type ReadAloudState = "idle" | "loading" | "speaking" | "paused";
3
4
  export interface ReadAloudChunk {
4
5
  /** The text of the chunk currently being spoken. */
@@ -22,6 +23,14 @@ export interface ReadAloudOptions {
22
23
  * seams between chunks more audible. Default 240.
23
24
  */
24
25
  maxChunkLength?: number;
26
+ /**
27
+ * How to read the text handed to `speak()`. `auto` (default) converts text
28
+ * that looks like Markdown so "##" and "**" are not read out; `markdown`
29
+ * always converts; `text` never does.
30
+ */
31
+ format?: "auto" | "markdown" | "text";
32
+ /** Markdown conversion options, when the text is read as Markdown. */
33
+ markdown?: MarkdownToSpeechOptions;
25
34
  /** Override synthesis entirely (tests, a bring-your-own-TTS host app). */
26
35
  synthesize?: SynthesizeFn;
27
36
  /** Called as each chunk starts playing. */
@@ -55,6 +64,11 @@ export declare class ReadAloudController {
55
64
  resume(): void;
56
65
  /** Stop playback and drop any queued chunks. Safe to call when already idle. */
57
66
  stop(): void;
67
+ /**
68
+ * Converts Markdown to spoken words before chunking, so a document read out
69
+ * of an editor does not have its syntax read back to the listener.
70
+ */
71
+ private toSpeakableText;
58
72
  private setState;
59
73
  private cleanup;
60
74
  private synthesize;
package/dist/client.js CHANGED
@@ -1,66 +1,29 @@
1
- var U = Object.defineProperty;
2
- var E = (h, e, i) => e in h ? U(h, e, { enumerable: !0, configurable: !0, writable: !0, value: i }) : h[e] = i;
3
- var l = (h, e, i) => E(h, typeof e != "symbol" ? e + "" : e, i);
4
- function C(h, e = 500) {
5
- const i = h.split(/\n\s*\n/), t = [];
6
- for (let s of i) {
7
- if (s.length <= e) {
8
- t.push(s.trim());
9
- continue;
10
- }
11
- const n = new RegExp(`(?<=[.?!])(?=\\s+["“”'a-z])`, "gi"), a = s.split(n);
12
- let o = "";
13
- for (let r of a) {
14
- if (r = r.trim(), r.length > e) {
15
- const p = k(r, e);
16
- for (let c of p)
17
- (o + " " + c).length > e ? (o && t.push(o.trim()), o = c) : o += (o ? " " : "") + c;
18
- continue;
19
- }
20
- (o + " " + r).length > e ? (o && t.push(o.trim()), o = r) : o += (o ? " " : "") + r;
21
- }
22
- o && t.push(o.trim());
23
- }
24
- return t;
25
- }
26
- function k(h, e) {
27
- const i = [];
28
- let t = "";
29
- const s = h.split(/,\s*/);
30
- for (let n of s)
31
- if ((t + ", " + n).length > e)
32
- if (t && i.push(t.trim()), n.length > e) {
33
- const a = n.split(/\s+/);
34
- let o = "";
35
- for (let r of a)
36
- (o + " " + r).length > e ? (o && i.push(o.trim()), o = r) : o += (o ? " " : "") + r;
37
- o && i.push(o.trim()), t = "";
38
- } else
39
- t = n;
40
- else
41
- t += (t ? ", " : "") + n;
42
- return t && i.push(t.trim()), i;
43
- }
44
- const v = "/api/speech/tts", R = "af_heart", L = 240;
45
- function w(h) {
46
- return h.aborted ? Promise.resolve() : new Promise((e) => {
47
- h.addEventListener("abort", () => e(), { once: !0 });
1
+ var k = Object.defineProperty;
2
+ var U = (r, t, e) => t in r ? k(r, t, { enumerable: !0, configurable: !0, writable: !0, value: e }) : r[t] = e;
3
+ var a = (r, t, e) => U(r, typeof t != "symbol" ? t + "" : t, e);
4
+ import { looksLikeMarkdown as E, markdownToSpeech as L } from "./markdown.js";
5
+ import { markdownToSpeechSegments as D, stripInlineMarkdown as N } from "./markdown.js";
6
+ import { s as C } from "./semantic-split-CXhk-k1F.js";
7
+ const T = "/api/speech/tts", v = "af_heart", R = 240;
8
+ function w(r) {
9
+ return r.aborted ? Promise.resolve() : new Promise((t) => {
10
+ r.addEventListener("abort", () => t(), { once: !0 });
48
11
  });
49
12
  }
50
- class A {
51
- constructor(e = {}) {
52
- l(this, "options");
53
- l(this, "state", "idle");
54
- l(this, "abortController", null);
55
- l(this, "audio", null);
56
- l(this, "objectUrl", null);
13
+ class O {
14
+ constructor(t = {}) {
15
+ a(this, "options");
16
+ a(this, "state", "idle");
17
+ a(this, "abortController", null);
18
+ a(this, "audio", null);
19
+ a(this, "objectUrl", null);
57
20
  /** Set once the endpoint has proved unreachable, so later chunks skip the retry. */
58
- l(this, "endpointUnavailable", !1);
59
- this.options = e;
21
+ a(this, "endpointUnavailable", !1);
22
+ this.options = t;
60
23
  }
61
24
  /** Replace the options (voice, callbacks, …) without discarding playback state. */
62
- setOptions(e) {
63
- this.options = e;
25
+ setOptions(t) {
26
+ this.options = t;
64
27
  }
65
28
  getState() {
66
29
  return this.state;
@@ -72,29 +35,29 @@ class A {
72
35
  * Speak `text`, cancelling anything already playing. Resolves when playback
73
36
  * finishes or is stopped — it never rejects; failures go to `onError`.
74
37
  */
75
- async speak(e) {
76
- var a, o, r, p, c, g, d, y;
38
+ async speak(t) {
39
+ var o, h, c, f, u, g, d, y;
77
40
  this.stop();
78
- const i = this.options.maxChunkLength ?? L, t = C(e ?? "", i).map((u) => u.trim()).filter((u) => u.length > 0);
79
- if (t.length === 0) return;
41
+ const e = this.options.maxChunkLength ?? R, i = C(this.toSpeakableText(t ?? ""), e).map((l) => l.trim()).filter((l) => l.length > 0);
42
+ if (i.length === 0) return;
80
43
  const s = new AbortController();
81
44
  this.abortController = s;
82
45
  const { signal: n } = s;
83
46
  this.setState("loading");
84
47
  try {
85
- let u = this.synthesize(t[0], n);
86
- for (let f = 0; f < t.length; f += 1) {
87
- const m = u;
88
- u = f + 1 < t.length ? this.synthesize(t[f + 1], n) : Promise.resolve(null);
89
- const S = await m;
90
- if (n.aborted || (this.setState("speaking"), (o = (a = this.options).onChunk) == null || o.call(a, { text: t[f], index: f, total: t.length }), n.aborted) || (await this.playChunk(S, t[f], n), n.aborted)) return;
48
+ let l = this.synthesize(i[0], n);
49
+ for (let p = 0; p < i.length; p += 1) {
50
+ const S = l;
51
+ l = p + 1 < i.length ? this.synthesize(i[p + 1], n) : Promise.resolve(null);
52
+ const b = await S;
53
+ if (n.aborted || (this.setState("speaking"), (h = (o = this.options).onChunk) == null || h.call(o, { text: i[p], index: p, total: i.length }), n.aborted) || (await this.playChunk(b, i[p], n), n.aborted)) return;
91
54
  }
92
- this.cleanup(), this.setState("idle"), (p = (r = this.options).onEnd) == null || p.call(r, "finished");
93
- } catch (u) {
55
+ this.cleanup(), this.setState("idle"), (f = (c = this.options).onEnd) == null || f.call(c, "finished");
56
+ } catch (l) {
94
57
  if (n.aborted) return;
95
- this.cleanup(), this.setState("idle"), (g = (c = this.options).onError) == null || g.call(
96
- c,
97
- u instanceof Error ? u : new Error(String(u))
58
+ this.cleanup(), this.setState("idle"), (g = (u = this.options).onError) == null || g.call(
59
+ u,
60
+ l instanceof Error ? l : new Error(String(l))
98
61
  ), (y = (d = this.options).onEnd) == null || y.call(d, "stopped");
99
62
  } finally {
100
63
  this.abortController === s && (this.abortController = null);
@@ -109,117 +72,125 @@ class A {
109
72
  }
110
73
  /** Stop playback and drop any queued chunks. Safe to call when already idle. */
111
74
  stop() {
112
- var i, t, s;
113
- const e = this.isActive();
114
- (i = this.abortController) == null || i.abort(), this.abortController = null, this.cleanup(), e && (this.setState("idle"), (s = (t = this.options).onEnd) == null || s.call(t, "stopped"));
75
+ var e, i, s;
76
+ const t = this.isActive();
77
+ (e = this.abortController) == null || e.abort(), this.abortController = null, this.cleanup(), t && (this.setState("idle"), (s = (i = this.options).onEnd) == null || s.call(i, "stopped"));
115
78
  }
116
- setState(e) {
117
- var i, t;
118
- this.state !== e && (this.state = e, (t = (i = this.options).onStateChange) == null || t.call(i, e));
79
+ /**
80
+ * Converts Markdown to spoken words before chunking, so a document read out
81
+ * of an editor does not have its syntax read back to the listener.
82
+ */
83
+ toSpeakableText(t) {
84
+ const e = this.options.format ?? "auto";
85
+ return e === "text" ? t : e === "markdown" || E(t) ? L(t, this.options.markdown) : t;
86
+ }
87
+ setState(t) {
88
+ var e, i;
89
+ this.state !== t && (this.state = t, (i = (e = this.options).onStateChange) == null || i.call(e, t));
119
90
  }
120
91
  cleanup() {
121
92
  this.audio && (this.audio.pause(), this.audio.src = "", this.audio = null), this.objectUrl && (URL.revokeObjectURL(this.objectUrl), this.objectUrl = null), typeof speechSynthesis < "u" && speechSynthesis.cancel();
122
93
  }
123
- synthesize(e, i) {
124
- const t = this.options.synthesize ? this.options.synthesize(e, i) : this.fetchAudio(e, i);
125
- return t.catch(() => {
126
- }), t;
94
+ synthesize(t, e) {
95
+ const i = this.options.synthesize ? this.options.synthesize(t, e) : this.fetchAudio(t, e);
96
+ return i.catch(() => {
97
+ }), i;
127
98
  }
128
- async fetchAudio(e, i) {
99
+ async fetchAudio(t, e) {
129
100
  if (this.endpointUnavailable) return null;
130
- const t = this.options.endpoint ?? v;
101
+ const i = this.options.endpoint ?? T;
131
102
  try {
132
- const s = await fetch(t, {
103
+ const s = await fetch(i, {
133
104
  method: "POST",
134
105
  headers: { "Content-Type": "application/json" },
135
106
  body: JSON.stringify({
136
- text: e,
107
+ text: t,
137
108
  provider: this.options.provider ?? "kokoro",
138
- voice: this.options.voice ?? R
109
+ voice: this.options.voice ?? v
139
110
  }),
140
- signal: i
111
+ signal: e
141
112
  });
142
113
  if (!s.ok) throw new Error(`TTS failed: ${s.status}`);
143
114
  const n = s.headers.get("Content-Type") || "audio/wav";
144
115
  return new Blob([await s.arrayBuffer()], { type: n });
145
116
  } catch (s) {
146
- if (i.aborted) throw s;
117
+ if (e.aborted) throw s;
147
118
  return this.endpointUnavailable = !0, null;
148
119
  }
149
120
  }
150
- playChunk(e, i, t) {
151
- return e ? this.playAudioBlob(e, t) : this.playWithSpeechSynthesis(i, t);
121
+ playChunk(t, e, i) {
122
+ return t ? this.playAudioBlob(t, i) : this.playWithSpeechSynthesis(e, i);
152
123
  }
153
- async playAudioBlob(e, i) {
154
- const t = URL.createObjectURL(e), s = new Audio(t);
155
- this.audio = s, this.objectUrl = t;
156
- const n = new Promise((a, o) => {
157
- s.addEventListener("ended", () => a(), { once: !0 }), s.addEventListener(
124
+ async playAudioBlob(t, e) {
125
+ const i = URL.createObjectURL(t), s = new Audio(i);
126
+ this.audio = s, this.objectUrl = i;
127
+ const n = new Promise((o, h) => {
128
+ s.addEventListener("ended", () => o(), { once: !0 }), s.addEventListener(
158
129
  "error",
159
- () => o(new Error("Audio playback failed")),
130
+ () => h(new Error("Audio playback failed")),
160
131
  { once: !0 }
161
132
  );
162
133
  });
163
134
  try {
164
- await s.play(), await Promise.race([n, w(i)]);
135
+ await s.play(), await Promise.race([n, w(e)]);
165
136
  } finally {
166
- this.audio === s && (this.audio = null), this.objectUrl === t && (this.objectUrl = null), s.pause(), URL.revokeObjectURL(t);
137
+ this.audio === s && (this.audio = null), this.objectUrl === i && (this.objectUrl = null), s.pause(), URL.revokeObjectURL(i);
167
138
  }
168
139
  }
169
- async playWithSpeechSynthesis(e, i) {
140
+ async playWithSpeechSynthesis(t, e) {
170
141
  if (typeof speechSynthesis > "u" || typeof SpeechSynthesisUtterance > "u")
171
142
  throw new Error("No speech synthesis available in this browser");
172
- const t = new SpeechSynthesisUtterance(e), s = new Promise((n, a) => {
173
- t.onend = () => n(), t.onerror = () => i.aborted ? n() : a(new Error("Speech synthesis failed"));
143
+ const i = new SpeechSynthesisUtterance(t), s = new Promise((n, o) => {
144
+ i.onend = () => n(), i.onerror = () => e.aborted ? n() : o(new Error("Speech synthesis failed"));
174
145
  });
175
- speechSynthesis.speak(t), await Promise.race([s, w(i)]);
146
+ speechSynthesis.speak(i), await Promise.race([s, w(e)]);
176
147
  }
177
148
  }
178
- function b() {
149
+ function m() {
179
150
  return typeof window > "u" ? null : window.SpeechRecognition || window.webkitSpeechRecognition || null;
180
151
  }
181
- function P() {
182
- var h;
183
- return typeof window > "u" ? !1 : b() ? !0 : !!((h = navigator.mediaDevices) != null && h.getUserMedia);
152
+ function x() {
153
+ var r;
154
+ return typeof window > "u" ? !1 : m() ? !0 : !!((r = navigator.mediaDevices) != null && r.getUserMedia);
184
155
  }
185
- class j {
186
- constructor(e = {}) {
187
- l(this, "options");
188
- l(this, "listening", !1);
156
+ class M {
157
+ constructor(t = {}) {
158
+ a(this, "options");
159
+ a(this, "listening", !1);
189
160
  /** Distinguishes a deliberate `stop()` from Chromium's idle auto-stop. */
190
- l(this, "stopRequested", !1);
191
- l(this, "recognition", null);
192
- l(this, "moonshine", null);
193
- this.options = e;
161
+ a(this, "stopRequested", !1);
162
+ a(this, "recognition", null);
163
+ a(this, "moonshine", null);
164
+ this.options = t;
194
165
  }
195
- setOptions(e) {
196
- this.options = e;
166
+ setOptions(t) {
167
+ this.options = t;
197
168
  }
198
169
  isListening() {
199
170
  return this.listening;
200
171
  }
201
172
  async start() {
202
- var t, s;
173
+ var i, s;
203
174
  if (this.listening) return;
204
175
  this.stopRequested = !1;
205
- const e = this.options.engine ?? "auto", i = b();
176
+ const t = this.options.engine ?? "auto", e = m();
206
177
  try {
207
- if (e !== "moonshine" && i) {
208
- this.startWebSpeech(i);
178
+ if (t !== "moonshine" && e) {
179
+ this.startWebSpeech(e);
209
180
  return;
210
181
  }
211
- if (e === "webspeech")
182
+ if (t === "webspeech")
212
183
  throw new Error("Speech recognition is not available in this browser");
213
184
  await this.startMoonshine();
214
185
  } catch (n) {
215
- this.setListening(!1), (s = (t = this.options).onError) == null || s.call(
216
- t,
186
+ this.setListening(!1), (s = (i = this.options).onError) == null || s.call(
187
+ i,
217
188
  n instanceof Error ? n : new Error(String(n))
218
189
  );
219
190
  }
220
191
  }
221
192
  async stop() {
222
- var e, i;
193
+ var t, e;
223
194
  if (this.stopRequested = !0, this.recognition) {
224
195
  try {
225
196
  this.recognition.stop();
@@ -229,7 +200,7 @@ class j {
229
200
  }
230
201
  if (this.moonshine) {
231
202
  try {
232
- await ((i = (e = this.moonshine).stop) == null ? void 0 : i.call(e));
203
+ await ((e = (t = this.moonshine).stop) == null ? void 0 : e.call(t));
233
204
  } catch {
234
205
  }
235
206
  this.moonshine = null;
@@ -239,54 +210,54 @@ class j {
239
210
  async toggle() {
240
211
  this.listening ? await this.stop() : await this.start();
241
212
  }
242
- setListening(e) {
243
- var i, t;
244
- this.listening !== e && (this.listening = e, (t = (i = this.options).onStateChange) == null || t.call(i, e));
213
+ setListening(t) {
214
+ var e, i;
215
+ this.listening !== t && (this.listening = t, (i = (e = this.options).onStateChange) == null || i.call(e, t));
245
216
  }
246
- startWebSpeech(e) {
247
- const i = new e();
248
- i.continuous = !0, i.interimResults = !0, i.lang = this.options.language ?? "en-US", i.onstart = () => this.setListening(!0), i.onresult = (t) => {
249
- var n, a, o, r, p;
217
+ startWebSpeech(t) {
218
+ const e = new t();
219
+ e.continuous = !0, e.interimResults = !0, e.lang = this.options.language ?? "en-US", e.onstart = () => this.setListening(!0), e.onresult = (i) => {
220
+ var n, o, h, c, f;
250
221
  let s = "";
251
- for (let c = t.resultIndex; c < t.results.length; c += 1) {
252
- const g = t.results[c], d = String(((n = g[0]) == null ? void 0 : n.transcript) ?? "").trim();
253
- d && (g.isFinal ? (o = (a = this.options).onCommit) == null || o.call(a, d) : s += `${s ? " " : ""}${d}`);
222
+ for (let u = i.resultIndex; u < i.results.length; u += 1) {
223
+ const g = i.results[u], d = String(((n = g[0]) == null ? void 0 : n.transcript) ?? "").trim();
224
+ d && (g.isFinal ? (h = (o = this.options).onCommit) == null || h.call(o, d) : s += `${s ? " " : ""}${d}`);
254
225
  }
255
- s && ((p = (r = this.options).onPartial) == null || p.call(r, s));
256
- }, i.onerror = (t) => {
257
- var n, a;
258
- const s = t == null ? void 0 : t.error;
259
- s === "no-speech" || s === "aborted" || (this.stopRequested = !0, (a = (n = this.options).onError) == null || a.call(n, new Error(`Speech recognition error: ${s}`)));
260
- }, i.onend = () => {
261
- if (this.stopRequested || this.recognition !== i) {
226
+ s && ((f = (c = this.options).onPartial) == null || f.call(c, s));
227
+ }, e.onerror = (i) => {
228
+ var n, o;
229
+ const s = i == null ? void 0 : i.error;
230
+ s === "no-speech" || s === "aborted" || (this.stopRequested = !0, (o = (n = this.options).onError) == null || o.call(n, new Error(`Speech recognition error: ${s}`)));
231
+ }, e.onend = () => {
232
+ if (this.stopRequested || this.recognition !== e) {
262
233
  this.recognition = null, this.setListening(!1);
263
234
  return;
264
235
  }
265
236
  try {
266
- i.start();
237
+ e.start();
267
238
  } catch {
268
239
  this.recognition = null, this.setListening(!1);
269
240
  }
270
- }, this.recognition = i, i.start();
241
+ }, this.recognition = e, e.start();
271
242
  }
272
243
  async startMoonshine() {
273
- const e = await import("@moonshine-ai/moonshine-js"), i = new e.MicrophoneTranscriber(
244
+ const t = await import("@moonshine-ai/moonshine-js"), e = new t.MicrophoneTranscriber(
274
245
  this.options.model ?? "model/small",
275
246
  {
276
- onTranscriptionUpdated: (t) => {
247
+ onTranscriptionUpdated: (i) => {
277
248
  var s, n;
278
- (n = (s = this.options).onPartial) == null || n.call(s, String(t ?? "").trim());
249
+ (n = (s = this.options).onPartial) == null || n.call(s, String(i ?? "").trim());
279
250
  },
280
- onTranscriptionCommitted: (t) => {
281
- var n, a, o, r;
282
- const s = String(t ?? "").trim();
283
- s ? (a = (n = this.options).onCommit) == null || a.call(n, s) : (r = (o = this.options).onPartial) == null || r.call(o, "");
251
+ onTranscriptionCommitted: (i) => {
252
+ var n, o, h, c;
253
+ const s = String(i ?? "").trim();
254
+ s ? (o = (n = this.options).onCommit) == null || o.call(n, s) : (c = (h = this.options).onPartial) == null || c.call(h, "");
284
255
  }
285
256
  },
286
257
  !1
287
258
  // streaming mode
288
259
  );
289
- if (this.moonshine = i, await i.start(), this.stopRequested) {
260
+ if (this.moonshine = e, await e.start(), this.stopRequested) {
290
261
  await this.stop();
291
262
  return;
292
263
  }
@@ -294,8 +265,12 @@ class j {
294
265
  }
295
266
  }
296
267
  export {
297
- j as LiveTranscriber,
298
- A as ReadAloudController,
299
- P as isTranscriptionSupported
268
+ M as LiveTranscriber,
269
+ O as ReadAloudController,
270
+ x as isTranscriptionSupported,
271
+ E as looksLikeMarkdown,
272
+ L as markdownToSpeech,
273
+ D as markdownToSpeechSegments,
274
+ N as stripInlineMarkdown
300
275
  };
301
276
  //# sourceMappingURL=client.js.map