echogarden 0.11.12 → 0.11.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/data/schemas/options.json +16 -0
  2. package/dist/api/Alignment.js +2 -2
  3. package/dist/api/Alignment.js.map +1 -1
  4. package/dist/api/Recognition.js +2 -2
  5. package/dist/api/Recognition.js.map +1 -1
  6. package/dist/api/Synthesis.js +5 -4
  7. package/dist/api/Synthesis.js.map +1 -1
  8. package/dist/api/Translation.js +2 -2
  9. package/dist/api/Translation.js.map +1 -1
  10. package/dist/audio/AudioUtilities.d.ts +1 -0
  11. package/dist/audio/AudioUtilities.js +25 -7
  12. package/dist/audio/AudioUtilities.js.map +1 -1
  13. package/dist/cli/CLI.js +2 -2
  14. package/dist/cli/CLI.js.map +1 -1
  15. package/dist/recognition/WhisperSTT.js +2 -2
  16. package/dist/recognition/WhisperSTT.js.map +1 -1
  17. package/dist/subtitles/Subtitles.d.ts +10 -7
  18. package/dist/subtitles/Subtitles.js +268 -207
  19. package/dist/subtitles/Subtitles.js.map +1 -1
  20. package/docs/Options.md +4 -2
  21. package/package.json +7 -6
  22. package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
  23. package/src/alignment/DTWSequenceAlignment.ts +121 -0
  24. package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
  25. package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
  26. package/src/alignment/SpeechAlignment.ts +488 -0
  27. package/src/api/API.ts +12 -0
  28. package/src/api/APIOptions.ts +15 -0
  29. package/src/api/Alignment.ts +329 -0
  30. package/src/api/Common.ts +16 -0
  31. package/src/api/Denoising.ts +120 -0
  32. package/src/api/LanguageDetection.ts +286 -0
  33. package/src/api/Recognition.ts +344 -0
  34. package/src/api/Synthesis.ts +1735 -0
  35. package/src/api/Translation.ts +143 -0
  36. package/src/api/Vad.ts +172 -0
  37. package/src/audio/AudioBufferConversion.ts +248 -0
  38. package/src/audio/AudioPlayer.ts +358 -0
  39. package/src/audio/AudioRecorder.ts +91 -0
  40. package/src/audio/AudioUtilities.ts +392 -0
  41. package/src/audio/SoxPath.ts +24 -0
  42. package/src/cli/CLI.ts +1360 -0
  43. package/src/cli/CLIConfigFile.ts +91 -0
  44. package/src/cli/CLILauncher.ts +26 -0
  45. package/src/cli/CLIOptionsSchema.ts +54 -0
  46. package/src/cli/CLIParser.ts +41 -0
  47. package/src/cli/CLIStarter.ts +40 -0
  48. package/src/codecs/FFMpegTranscoder.ts +214 -0
  49. package/src/codecs/TIMITCodec.ts +17 -0
  50. package/src/codecs/WaveCodec.ts +260 -0
  51. package/src/denoising/RNNoise.ts +95 -0
  52. package/src/dsp/BiquadFilter.ts +488 -0
  53. package/src/dsp/FFT.ts +187 -0
  54. package/src/dsp/MFCC.ts +227 -0
  55. package/src/dsp/MelSpectogram.ts +145 -0
  56. package/src/dsp/Rubberband.ts +249 -0
  57. package/src/dsp/Sonic.ts +59 -0
  58. package/src/dsp/SpeexResampler.ts +79 -0
  59. package/src/math/VectorMath.ts +812 -0
  60. package/src/nlp/ChineseSegmentation.ts +68 -0
  61. package/src/nlp/CompromiseNLP.ts +113 -0
  62. package/src/nlp/EspeakPhonemizer.ts +168 -0
  63. package/src/nlp/IPA.ts +139 -0
  64. package/src/nlp/JapaneseSegmentation.ts +53 -0
  65. package/src/nlp/Lexicon.ts +119 -0
  66. package/src/nlp/PhoneConversion.ts +508 -0
  67. package/src/nlp/Segmentation.ts +237 -0
  68. package/src/nlp/TextNormalizer.ts +160 -0
  69. package/src/recognition/AmazonTranscribeSTT.ts +112 -0
  70. package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
  71. package/src/recognition/GoogleCloudSTT.ts +92 -0
  72. package/src/recognition/SileroSTT.ts +173 -0
  73. package/src/recognition/VoskSTT.ts +112 -0
  74. package/src/recognition/WhisperSTT.ts +1518 -0
  75. package/src/server/Client.ts +297 -0
  76. package/src/server/Server.ts +178 -0
  77. package/src/server/ServerStarter.ts +12 -0
  78. package/src/server/Worker.ts +400 -0
  79. package/src/server/WorkerStarter.ts +38 -0
  80. package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
  81. package/src/subtitles/Subtitles.ts +478 -0
  82. package/src/synthesis/AwsPollyTTS.ts +78 -0
  83. package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
  84. package/src/synthesis/CoquiServerTTS.ts +29 -0
  85. package/src/synthesis/ElevenLabsTTS.ts +104 -0
  86. package/src/synthesis/EspeakTTS.ts +552 -0
  87. package/src/synthesis/FliteTTS.ts +387 -0
  88. package/src/synthesis/GoogleCloudTTS.ts +112 -0
  89. package/src/synthesis/GoogleTranslateTTS.ts +210 -0
  90. package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
  91. package/src/synthesis/SamTTS.ts +30 -0
  92. package/src/synthesis/SapiTTS.ts +222 -0
  93. package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
  94. package/src/synthesis/SvoxPicoTTS.ts +318 -0
  95. package/src/synthesis/VitsTTS.ts +734 -0
  96. package/src/tests/Test.ts +24 -0
  97. package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
  98. package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
  99. package/src/typings/Fillers.d.ts +41 -0
  100. package/src/utilities/BinaryArrayConversion.ts +159 -0
  101. package/src/utilities/Compression.ts +91 -0
  102. package/src/utilities/FileDownloader.ts +201 -0
  103. package/src/utilities/FileSystem.ts +265 -0
  104. package/src/utilities/Hashing.ts +230 -0
  105. package/src/utilities/Locale.ts +119 -0
  106. package/src/utilities/Logger.ts +72 -0
  107. package/src/utilities/NdArrayUtilities.ts +31 -0
  108. package/src/utilities/ObjectUtilities.ts +169 -0
  109. package/src/utilities/OpenPromise.ts +13 -0
  110. package/src/utilities/PackageManager.ts +97 -0
  111. package/src/utilities/Queue.ts +17 -0
  112. package/src/utilities/RandomGenerator.ts +237 -0
  113. package/src/utilities/SignalChannel.ts +22 -0
  114. package/src/utilities/TarballMaker.ts +68 -0
  115. package/src/utilities/Timeline.ts +231 -0
  116. package/src/utilities/Timer.ts +93 -0
  117. package/src/utilities/Utilities.ts +574 -0
  118. package/src/utilities/WasmMemoryManager.ts +516 -0
  119. package/src/utilities/WebReader.ts +55 -0
  120. package/src/utilities/WikipediaReader.ts +41 -0
  121. package/src/voice-activity-detection/SileroVAD.ts +86 -0
  122. package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
package/src/cli/CLI.ts ADDED
@@ -0,0 +1,1360 @@
1
+ import * as API from '../api/API.js'
2
+ import { CLIArguments, parseCLIArguments } from './CLIParser.js'
3
+ import { convertHtmlToText, formatIntegerWithLeadingZeros, formatListWithQuotedElements, getWithDefault, logToStderr, setupUnhandledExceptionListeners, splitFilenameOnExtendedExtension, stringifyAndFormatJson } from "../utilities/Utilities.js"
4
+ import { getOptionTypeFromSchema, SchemaTypeDefinition } from "./CLIOptionsSchema.js"
5
+ import { ParsedConfigFile, parseConfigFile, parseJSONConfigFile } from "./CLIConfigFile.js"
6
+
7
+ import chalk from 'chalk'
8
+ import { RawAudio, applyGainDecibels, encodeWaveBuffer, getEmptyRawAudio, normalizeAudioLevel, sliceRawAudioByTime } from "../audio/AudioUtilities.js"
9
+ import { SubtitlesConfig, subtitlesToText, subtitlesToTimeline, timelineToSubtitles } from "../subtitles/Subtitles.js"
10
+ import { Logger, resetActiveLogger } from "../utilities/Logger.js"
11
+ import { isMainThread, parentPort } from 'node:worker_threads'
12
+ import { encodeFromChannels, getDefaultFFMpegOptionsForSpeech } from "../codecs/FFMpegTranscoder.js"
13
+ import path, { parse as parsePath } from "node:path"
14
+ import { splitToParagraphs, splitToWords, wordCharacterPattern } from "../nlp/Segmentation.js"
15
+ import { playAudioSamples, playAudioWithWordTimeline } from "../audio/AudioPlayer.js"
16
+ import { extendDeep } from "../utilities/ObjectUtilities.js"
17
+ import { Timeline, TimelineEntry, addTimeOffsetToTimeline, addWordTextOffsetsToTimeline, roundTimelineProperties, wordTimelineToSegmentSentenceTimeline } from "../utilities/Timeline.js"
18
+ import { ensureDir, existsSync, getLowercaseFileExtension, readAndParseJsonFile, readFile, readdir, resolveToModuleRootDir, writeFileSafe } from '../utilities/FileSystem.js'
19
+ import { formatLanguageCodeWithName, getShortLanguageCode } from '../utilities/Locale.js'
20
+ import { APIOptions } from '../api/APIOptions.js'
21
+ import { ensureAndGetPackagesDir, getVersionTagFromPackageName, loadPackage, resolveVersionTagForUnversionedPackageName } from '../utilities/PackageManager.js'
22
+ import { removePackage } from '../utilities/PackageManager.js'
23
+ import { appName } from '../api/Common.js'
24
+ import { ServerOptions, startServer } from '../server/Server.js'
25
+ import { OpenPromise } from '../utilities/OpenPromise.js'
26
+ import JSON5 from 'json5'
27
+
28
+ //const log = logToStderr
29
+
30
+ async function startIfInWorkerThread() {
31
+ if (isMainThread || !parentPort) {
32
+ return
33
+ }
34
+
35
+ setupUnhandledExceptionListeners()
36
+
37
+ const initOpenPromise = new OpenPromise<void>()
38
+
39
+ parentPort.once("message", (message) => {
40
+ if (message.name == 'init') {
41
+ process.stderr.isTTY = message.stdErrIsTTY
42
+ process.stderr.hasColors = () => message.hasColors
43
+
44
+ process.stderr.write = (text) => {
45
+ parentPort!.postMessage({ name: 'writeToStdErr', text })
46
+ return true
47
+ }
48
+
49
+ initOpenPromise.resolve()
50
+ }
51
+ })
52
+
53
+ await initOpenPromise.promise
54
+
55
+ start(process.argv.slice(2))
56
+ }
57
+
58
+ export async function start(processArgs: string[]) {
59
+ const logger = new Logger()
60
+
61
+ let args: CLIArguments
62
+
63
+ try {
64
+ const packageData = await readAndParseJsonFile(resolveToModuleRootDir("package.json"))
65
+
66
+ logger.log(chalk.magentaBright(`Echogarden v${packageData.version}\n`))
67
+
68
+ const command = processArgs[0]
69
+
70
+ if (!command || command == "help") {
71
+ logger.log(`Supported operations:\n\n${commandHelp.join("\n")}`)
72
+ process.exit(0)
73
+ }
74
+
75
+ if (command == "--help") {
76
+ logger.log(`There's no command called '--help'. Did you mean to run 'echogarden help'?`)
77
+ process.exit(0)
78
+ }
79
+
80
+ args = parseCLIArguments(command, processArgs.slice(1))
81
+
82
+ if (!args.options.has("config")) {
83
+ const defaultConfigFile = `./${appName}.config`
84
+ const defaultJsonConfigFile = defaultConfigFile + ".json"
85
+
86
+ if (existsSync(defaultConfigFile)) {
87
+ args.options.set("config", defaultConfigFile)
88
+ } else if (existsSync(defaultJsonConfigFile)) {
89
+ args.options.set("config", defaultJsonConfigFile)
90
+ }
91
+ }
92
+
93
+ if (args.options.has("config")) {
94
+ const configFilePath = args.options.get("config")!
95
+ args.options.delete("config")
96
+
97
+ let parsedOptionFile: ParsedConfigFile
98
+
99
+ if (configFilePath.endsWith(".config")) {
100
+ parsedOptionFile = await parseConfigFile(configFilePath)
101
+ } else if (configFilePath.endsWith(".config.json")) {
102
+ parsedOptionFile = await parseJSONConfigFile(configFilePath)
103
+ } else {
104
+ throw new Error(`Specified config file '${configFilePath}' doesn't have a supported extension. Should be either '.config' or '.config.json'`)
105
+ }
106
+
107
+ let sectionName = args.command
108
+
109
+ if (sectionName.startsWith("speak-")) {
110
+ sectionName = "speak"
111
+ }
112
+
113
+ const newOptions = parsedOptionFile.get(sectionName) || new Map<string, string>()
114
+
115
+ for (const [key, value] of args.options) {
116
+ newOptions.set(key, value)
117
+ }
118
+
119
+ args.options = newOptions
120
+ }
121
+ } catch (e: any) {
122
+ resetActiveLogger()
123
+
124
+ logger.logTitledMessage(`Error`, e.message, chalk.redBright)
125
+ process.exit(1)
126
+ }
127
+
128
+ let debugMode = false
129
+ if (args.options.has('debug')) {
130
+ args.options.delete('debug')
131
+
132
+ debugMode = true
133
+ }
134
+
135
+ try {
136
+ await startWithArgs(args)
137
+ } catch (e: any) {
138
+ resetActiveLogger()
139
+
140
+ if (debugMode) {
141
+ logger.log(e)
142
+ } else {
143
+ logger.logTitledMessage(`Error`, e.message, chalk.redBright)
144
+ }
145
+
146
+ process.exit(1)
147
+ }
148
+
149
+ process.exit(0)
150
+ }
151
+
152
+ const executableName = `${chalk.cyanBright('echogarden')}`
153
+
154
+ const commandHelp = [
155
+ `${executableName} ${chalk.magentaBright('speak')} text [output files...] [options...]`,
156
+ ` Speak the given text\n`,
157
+ `${executableName} ${chalk.magentaBright('speak-file')} inputFile [output files...] [options...]`,
158
+ ` Speak the given text file\n`,
159
+ `${executableName} ${chalk.magentaBright('speak-url')} url [output files...] [options...]`,
160
+ ` Speak the HTML document on the given URL\n`,
161
+ `${executableName} ${chalk.magentaBright('speak-wikipedia')} articleName [output files...] [options...]`,
162
+ ` Speak the given wikipedia article, language edition can be specified by --language=<langCode>\n`,
163
+ `${executableName} ${chalk.magentaBright('transcribe')} audioFile [output files...] [options...]`,
164
+ ` Transcribe audio file\n`,
165
+ `${executableName} ${chalk.magentaBright('align')} audioFile referenceFile [output files...] [options...]`,
166
+ ` Align audio file to the reference transcript file\n`,
167
+ `${executableName} ${chalk.magentaBright('translate-speech')} inputFile [output files...] [options...]`,
168
+ ` Transcribe audio file directly to a different language\n`,
169
+ `${executableName} ${chalk.magentaBright('detect-speech-language')} audioFile [output files...] [options...]`,
170
+ ` Detect language of audio file\n`,
171
+ `${executableName} ${chalk.magentaBright('detect-text-language')} inputFile [output files...] [options...]`,
172
+ ` Detect language of textual file\n`,
173
+ `${executableName} ${chalk.magentaBright('detect-voice-activity')} audioFile [output files...] [options...]`,
174
+ ` Detect voice activity in audio file\n`,
175
+ `${executableName} ${chalk.magentaBright('denoise')} audioFile [output files...] [options...]`,
176
+ ` Denoise audio file\n`,
177
+ `${executableName} ${chalk.magentaBright('list-engines')} operation`,
178
+ ` List available engines for the specified operation\n`,
179
+ `${executableName} ${chalk.magentaBright('list-voices')} tts-engine [output files...] [options...]`,
180
+ ` List available voices for the specified TTS engine\n`,
181
+ `${executableName} ${chalk.magentaBright('install')} [package names...] [options...]`,
182
+ ` Install one or more Echogarden packages\n`,
183
+ `${executableName} ${chalk.magentaBright('uninstall')} [package names...] [options...]`,
184
+ ` Uninstall one or more Echogarden packages\n`,
185
+ `${executableName} ${chalk.magentaBright('list-packages')} [options...]`,
186
+ ` List installed Echogarden packages\n`,
187
+ `${executableName} ${chalk.magentaBright('serve')} [options...]`,
188
+ ` Start a server\n`,
189
+ ]
190
+
191
+ async function startWithArgs(parsedArgs: CLIArguments) {
192
+ const logger = new Logger()
193
+
194
+ switch (parsedArgs.command) {
195
+ case 'speak':
196
+ case 'speak-file':
197
+ case 'speak-url':
198
+ case 'speak-wikipedia': {
199
+ await speak(parsedArgs.command, parsedArgs.commandArgs, parsedArgs.options)
200
+ break
201
+ }
202
+
203
+ case 'transcribe': {
204
+ await transcribe(parsedArgs.commandArgs, parsedArgs.options)
205
+ break
206
+ }
207
+
208
+ case 'align': {
209
+ await align(parsedArgs.commandArgs, parsedArgs.options)
210
+ break
211
+ }
212
+
213
+ case 'translate-speech': {
214
+ await translateSpeech(parsedArgs.commandArgs, parsedArgs.options)
215
+ break
216
+ }
217
+
218
+ case 'detect-language': {
219
+ await detectLanguage(parsedArgs.commandArgs, parsedArgs.options, "auto")
220
+ break
221
+ }
222
+
223
+ case 'detect-speech-language': {
224
+ await detectLanguage(parsedArgs.commandArgs, parsedArgs.options, "speech")
225
+ break
226
+ }
227
+
228
+ case 'detect-text-language': {
229
+ await detectLanguage(parsedArgs.commandArgs, parsedArgs.options, "text")
230
+ break
231
+ }
232
+
233
+ case 'detect-voice-activity': {
234
+ await detectVoiceActivity(parsedArgs.commandArgs, parsedArgs.options)
235
+ break
236
+ }
237
+
238
+ case 'denoise': {
239
+ await denoise(parsedArgs.commandArgs, parsedArgs.options)
240
+ break
241
+ }
242
+
243
+ case 'list-engines': {
244
+ await listEngines(parsedArgs.commandArgs, parsedArgs.options)
245
+ break
246
+ }
247
+
248
+ case 'list-voices': {
249
+ await listTTSVoices(parsedArgs.commandArgs, parsedArgs.options)
250
+ break
251
+ }
252
+
253
+ case 'install': {
254
+ await installPackages(parsedArgs.commandArgs, parsedArgs.options)
255
+ break
256
+ }
257
+
258
+ case 'uninstall': {
259
+ await uninstallPackages(parsedArgs.commandArgs, parsedArgs.options)
260
+ break
261
+ }
262
+
263
+ case 'list-packages': {
264
+ await listPackages(parsedArgs.commandArgs, parsedArgs.options)
265
+ break
266
+ }
267
+
268
+ case 'serve': {
269
+ await serve(parsedArgs.commandArgs, parsedArgs.options)
270
+ break
271
+ }
272
+
273
+ default: {
274
+ logger.logTitledMessage(`Unknown command`, parsedArgs.command, chalk.redBright)
275
+ process.exit(1)
276
+ }
277
+ }
278
+ }
279
+
280
+ type SpeakCommand = "speak" | "speak-file" | "speak-url" | "speak-wikipedia"
281
+
282
+ async function speak(command: SpeakCommand, commandArgs: string[], cliOptions: Map<string, string>) {
283
+ const logger = new Logger()
284
+
285
+ const mainArg = commandArgs[0]
286
+ const outputFilenames = commandArgs.slice(1)
287
+
288
+ if (mainArg == undefined) {
289
+ if (command == "speak") {
290
+ throw new Error(`'speak' requires an argument containing the text to speak.`)
291
+ } else if (command == "speak-file") {
292
+ throw new Error(`'speak-file' requires an argument containing the file to speak.`)
293
+ } else if (command == "speak-url") {
294
+ throw new Error(`'speak-url' requires an argument containing the url to speak.`)
295
+ } else if (command == "speak-wikipedia") {
296
+ throw new Error(`'speak-wikipedia' requires an argument containing the name of the Wikipedia article to speak.`)
297
+ }
298
+
299
+ return
300
+ }
301
+
302
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
303
+ additionalOptionsSchema.set('play', { type: 'boolean' })
304
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
305
+
306
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
307
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
308
+ }
309
+
310
+ const options: API.SynthesisOptions = await cliOptionsMapToOptionsObject(cliOptions, "SynthesisOptions", additionalOptionsSchema)
311
+
312
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
313
+ const { includesPartPattern } = await checkOutputFilenames(outputFilenames, true, true, true)
314
+
315
+ let plainText: string | undefined = undefined
316
+ let textSegments: string[]
317
+
318
+ const plainTextParagraphBreaks = options.plainText?.paragraphBreaks || API.defaultSynthesisOptions.plainText!.paragraphBreaks!
319
+ const plainTextWhitespace = options.plainText?.whitespace || API.defaultSynthesisOptions.plainText!.whitespace!
320
+
321
+ if (command == "speak") {
322
+ if (options.ssml) {
323
+ textSegments = [mainArg]
324
+ } else {
325
+ textSegments = splitToParagraphs(mainArg, plainTextParagraphBreaks, plainTextWhitespace)
326
+ }
327
+
328
+ plainText = mainArg
329
+ } else if (command == "speak-file") {
330
+ const sourceFile = mainArg
331
+
332
+ if (!existsSync(sourceFile)) {
333
+ throw new Error(`The given source file '${sourceFile}' was not found.`)
334
+ }
335
+
336
+ const sourceFileExtension = getLowercaseFileExtension(sourceFile)
337
+ const fileContent = await readFile(sourceFile, { encoding: 'utf-8' })
338
+
339
+ if (options.ssml && sourceFileExtension != "xml" && sourceFileExtension != "ssml") {
340
+ throw new Error(`SSML option is set, but source file doesn't have an 'xml' or 'ssml' extension.`)
341
+ }
342
+
343
+ if (sourceFileExtension == "txt") {
344
+ textSegments = splitToParagraphs(fileContent, plainTextParagraphBreaks, plainTextWhitespace)
345
+
346
+ plainText = fileContent
347
+ } else if (sourceFileExtension == "html" || sourceFileExtension == "htm") {
348
+ const textContent = await convertHtmlToText(fileContent)
349
+ textSegments = splitToParagraphs(textContent, 'single', 'preserve')
350
+ } else if (sourceFileExtension == "srt" || sourceFileExtension == "vtt") {
351
+ const fileContent = await readFile(sourceFile, { encoding: 'utf-8' })
352
+ textSegments = subtitlesToTimeline(fileContent).map(entry => entry.text)
353
+ } else if (sourceFileExtension == "xml" || sourceFileExtension == "ssml") {
354
+ options.ssml = true
355
+ textSegments = [fileContent]
356
+ } else {
357
+ throw new Error(`'speak-file' only supports inputs with extensions 'txt', 'html', 'htm', 'xml', 'ssml', 'srt', 'vtt'`)
358
+ }
359
+ } else if (command == "speak-url") {
360
+ if (options.ssml) {
361
+ throw new Error(`speak-url doesn't provide SSML inputs`)
362
+ }
363
+
364
+ const url = mainArg
365
+
366
+ if (!url.startsWith("http://") && !url.startsWith("https://")) {
367
+ throw new Error(`'${url}' is not a valid URL. Only 'http://' and 'https://' protocols are supported`)
368
+ }
369
+
370
+ const { fetchDocumentText } = await import("../utilities/WebReader.js")
371
+ const textContent = await fetchDocumentText(url)
372
+
373
+ textSegments = splitToParagraphs(textContent, 'single', 'preserve')
374
+ } else if (command == "speak-wikipedia") {
375
+ if (options.ssml) {
376
+ throw new Error(`speak-wikipedia doesn't provide SSML inputs`)
377
+ }
378
+
379
+ const { parseWikipediaArticle } = await import("../utilities/WikipediaReader.js")
380
+ if (!options.language) {
381
+ options.language = "en"
382
+ }
383
+
384
+ textSegments = await parseWikipediaArticle(mainArg, getShortLanguageCode(options.language))
385
+ } else {
386
+ throw new Error("Invalid command")
387
+ }
388
+
389
+ async function onSegment(segmentData: API.SynthesisSegmentEventData) {
390
+ if (includesPartPattern) {
391
+ logger.start("Writing output files for segment")
392
+ }
393
+
394
+ await writeOutputFilesForSegment(outputFilenames, segmentData.index, segmentData.total, segmentData.audio as RawAudio, segmentData.timeline, segmentData.transcript, segmentData.language, allowOverwrite)
395
+
396
+ logger.end()
397
+
398
+ if ((options as any).play) {
399
+ let gainAmount = -3 - segmentData.peakDecibelsSoFar
400
+ //gainAmount = Math.min(gainAmount, 0)
401
+
402
+ const audioWithAddedGain = applyGainDecibels(segmentData.audio as RawAudio, gainAmount)
403
+ const segmentWordTimeline = segmentData.timeline.flatMap(sentenceTimeline => sentenceTimeline.timeline!)
404
+
405
+ await playAudioWithWordTimeline(audioWithAddedGain, segmentWordTimeline, segmentData.transcript)
406
+ }
407
+ }
408
+
409
+ if (options.outputAudioFormat?.codec) {
410
+ options.outputAudioFormat!.codec = undefined
411
+ }
412
+
413
+ const { audio: synthesizedAudio, timeline } = await API.synthesize(textSegments, options, onSegment, undefined)
414
+
415
+ if (plainText) {
416
+ addWordTextOffsetsToTimeline(timeline, plainText)
417
+ }
418
+
419
+ if (outputFilenames.length > 0) {
420
+ logger.start("\nWriting output files")
421
+ }
422
+
423
+ for (const outputFile of outputFilenames) {
424
+ const partPatternMatch = outputFile.match(segmentFilenamePattern)
425
+
426
+ if (partPatternMatch) {
427
+ continue
428
+ }
429
+
430
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
431
+ await fileSaver(synthesizedAudio as RawAudio, timeline, textSegments.join("\n\n"), options.subtitles)
432
+ }
433
+
434
+ logger.end()
435
+ }
436
+
437
+ async function transcribe(commandArgs: string[], cliOptions: Map<string, string>) {
438
+ const logger = new Logger()
439
+
440
+ const sourceFilename = commandArgs[0]
441
+ const outputFilenames = commandArgs.slice(1)
442
+
443
+ if (sourceFilename == undefined) {
444
+ throw new Error(`'transcribe' requires an argument containing the source file name.`)
445
+ }
446
+
447
+ if (!existsSync(sourceFilename)) {
448
+ throw new Error(`The given source audio file '${sourceFilename}' was not found.`)
449
+ }
450
+
451
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
452
+ additionalOptionsSchema.set('play', { type: 'boolean' })
453
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
454
+
455
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
456
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
457
+ }
458
+
459
+ const options: API.RecognitionOptions = await cliOptionsMapToOptionsObject(cliOptions, "RecognitionOptions", additionalOptionsSchema)
460
+
461
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
462
+ const { includesPartPattern } = await checkOutputFilenames(outputFilenames, true, true, true)
463
+
464
+ const { transcript, timeline, wordTimeline, inputRawAudio, language } = await API.recognize(sourceFilename, options)
465
+
466
+ if (outputFilenames.length > 0) {
467
+ logger.start("\nWriting output files")
468
+ }
469
+
470
+ for (const outputFile of outputFilenames) {
471
+ const partPatternMatch = outputFile.match(segmentFilenamePattern)
472
+
473
+ if (partPatternMatch) {
474
+ continue
475
+ }
476
+
477
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
478
+
479
+ await fileSaver(inputRawAudio, timeline, transcript, options.subtitles)
480
+ }
481
+
482
+ logger.end()
483
+
484
+ if ((options as any).play) {
485
+ const normalizedAudio = normalizeAudioLevel(inputRawAudio)
486
+
487
+ await playAudioWithWordTimeline(normalizedAudio, wordTimeline, transcript)
488
+ }
489
+ }
490
+
491
+ async function align(commandArgs: string[], cliOptions: Map<string, string>) {
492
+ const logger = new Logger()
493
+
494
+ const audioFilename = commandArgs[0]
495
+ const outputFilenames = commandArgs.slice(2)
496
+
497
+ if (audioFilename == undefined) {
498
+ throw new Error(`align requires an argument containing the audio file path.`)
499
+ }
500
+
501
+ if (!existsSync(audioFilename)) {
502
+ throw new Error(`The given source file '${audioFilename}' was not found.`)
503
+ }
504
+
505
+ const alignmentReferenceFile = commandArgs[1]
506
+
507
+ if (alignmentReferenceFile == undefined) {
508
+ throw new Error(`align requires a second argument containing the alignment reference file path.`)
509
+ }
510
+
511
+ if (!existsSync(alignmentReferenceFile)) {
512
+ throw new Error(`The given reference file '${alignmentReferenceFile}' was not found.`)
513
+ }
514
+
515
+ const referenceFileExtension = getLowercaseFileExtension(alignmentReferenceFile)
516
+ const fileContent = await readFile(alignmentReferenceFile, { encoding: 'utf-8' })
517
+
518
+ let text: string
519
+
520
+ if (referenceFileExtension == "txt") {
521
+ text = fileContent
522
+ } else if (referenceFileExtension == "html" || referenceFileExtension == "htm") {
523
+ text = await convertHtmlToText(fileContent)
524
+ } else if (referenceFileExtension == "srt" || referenceFileExtension == "vtt") {
525
+ text = subtitlesToText(fileContent)
526
+ } else {
527
+ throw new Error(`align only supports reference files with extensions 'txt', 'html', 'htm', 'srt' or 'vtt'`)
528
+ }
529
+
530
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
531
+ additionalOptionsSchema.set('play', { type: 'boolean' })
532
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
533
+
534
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
535
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
536
+ }
537
+
538
+ const options: API.AlignmentOptions = await cliOptionsMapToOptionsObject(cliOptions, "AlignmentOptions", additionalOptionsSchema)
539
+
540
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
541
+ const { includesPartPattern } = await checkOutputFilenames(outputFilenames, true, true, true)
542
+
543
+ const { timeline, wordTimeline, transcript, language, inputRawAudio } = await API.align(audioFilename, text, options)
544
+
545
+ if (outputFilenames.length > 0) {
546
+ logger.start("\nWriting output files")
547
+ }
548
+
549
+ if (includesPartPattern) {
550
+ for (let segmentIndex = 0; segmentIndex < timeline.length; segmentIndex++) {
551
+ const segmentEntry = timeline[segmentIndex]
552
+ const segmentAudio = sliceRawAudioByTime(inputRawAudio, segmentEntry.startTime, segmentEntry.endTime)
553
+ const sentenceTimeline = addTimeOffsetToTimeline(segmentEntry.timeline!, -segmentEntry.startTime)
554
+
555
+ await writeOutputFilesForSegment(outputFilenames, segmentIndex, timeline.length, segmentAudio, sentenceTimeline, segmentEntry.text, language, allowOverwrite)
556
+ }
557
+ }
558
+
559
+ for (const outputFile of outputFilenames) {
560
+ const partPatternMatch = outputFile.match(segmentFilenamePattern)
561
+
562
+ if (partPatternMatch) {
563
+ continue
564
+ }
565
+
566
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
567
+
568
+ await fileSaver(inputRawAudio, timeline, transcript, options.subtitles)
569
+ }
570
+
571
+ logger.end()
572
+
573
+ if ((options as any).play) {
574
+ const normalizedAudio = normalizeAudioLevel(inputRawAudio)
575
+
576
+ await playAudioWithWordTimeline(normalizedAudio, wordTimeline, transcript)
577
+ }
578
+ }
579
+
580
+ async function translateSpeech(commandArgs: string[], cliOptions: Map<string, string>) {
581
+ const logger = new Logger()
582
+
583
+ const inputFilename = commandArgs[0]
584
+ const outputFilenames = commandArgs.slice(1)
585
+
586
+ if (inputFilename == undefined) {
587
+ throw new Error(`translate-speech requires an argument containing the input file path.`)
588
+ }
589
+
590
+ if (!existsSync(inputFilename)) {
591
+ throw new Error(`The given input file '${inputFilename}' was not found.`)
592
+ }
593
+
594
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
595
+ additionalOptionsSchema.set('play', { type: 'boolean' })
596
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
597
+
598
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
599
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
600
+ }
601
+
602
+ const options: API.SpeechTranslationOptions = await cliOptionsMapToOptionsObject(cliOptions, "SpeechTranslationOptions", additionalOptionsSchema)
603
+
604
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
605
+
606
+ await checkOutputFilenames(outputFilenames, true, true, true)
607
+
608
+ const { transcript, timeline, wordTimeline, sourceLanguage, targetLanguage, inputRawAudio } = await API.translateSpeech(inputFilename, options)
609
+
610
+ if (outputFilenames.length > 0) {
611
+ logger.start("\nWriting output files")
612
+ }
613
+
614
+ for (const outputFile of outputFilenames) {
615
+ const partPatternMatch = outputFile.match(segmentFilenamePattern)
616
+
617
+ if (partPatternMatch) {
618
+ continue
619
+ }
620
+
621
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
622
+
623
+ await fileSaver(inputRawAudio, timeline, transcript, options.subtitles)
624
+ }
625
+
626
+ logger.end()
627
+
628
+ if ((options as any).play) {
629
+ const normalizedAudio = normalizeAudioLevel(inputRawAudio)
630
+
631
+ await playAudioWithWordTimeline(normalizedAudio, wordTimeline, transcript)
632
+ }
633
+ }
634
+
635
+ async function detectLanguage(commandArgs: string[], cliOptions: Map<string, string>, mode: "speech" | "text" | "auto") {
636
+ const logger = new Logger()
637
+
638
+ const inputFilePath = commandArgs[0]
639
+ const outputFilenames = commandArgs.slice(1)
640
+
641
+ if (!existsSync(inputFilePath)) {
642
+ throw new Error(`The given input file '${inputFilePath}' was not found.`)
643
+ }
644
+
645
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
646
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
647
+
648
+ const inputFileExtension = getLowercaseFileExtension(inputFilePath)
649
+ const supportedInputTextFormats = ["txt", "srt", "vtt"]
650
+
651
+ let results: API.LanguageDetectionResults
652
+
653
+ let allowOverwrite: boolean
654
+
655
+ if (mode == "text" || (mode == "auto" && supportedInputTextFormats.includes(inputFileExtension))) {
656
+ if (inputFilePath == undefined) {
657
+ throw new Error(`detect-text-language requires an argument containing the input file path.`)
658
+ }
659
+
660
+ if (!supportedInputTextFormats.includes(inputFileExtension)) {
661
+ throw new Error(`'detect-text-language' doesn't support input file extension '${inputFileExtension}'`)
662
+ }
663
+
664
+ const options: API.TextLanguageDetectionOptions = await cliOptionsMapToOptionsObject(cliOptions, "TextLanguageDetectionOptions", additionalOptionsSchema)
665
+ allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
666
+
667
+ await checkOutputFilenames(outputFilenames, false, true, false)
668
+
669
+ let text = await readFile(inputFilePath, { encoding: "utf-8" })
670
+
671
+ if (inputFileExtension == "srt" || inputFileExtension == "vtt") {
672
+ text = subtitlesToText(text)
673
+ }
674
+
675
+ const { detectedLanguage, detectedLanguageProbabilities } = await API.detectTextLanguage(text, options)
676
+
677
+ results = detectedLanguageProbabilities
678
+ } else {
679
+ if (inputFilePath == undefined) {
680
+ throw new Error(`detect-speech-language requires an argument containing the input audio file path.`)
681
+ }
682
+
683
+ const options: API.SpeechLanguageDetectionOptions = await cliOptionsMapToOptionsObject(cliOptions, "SpeechLanguageDetectionOptions", additionalOptionsSchema)
684
+ allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
685
+
686
+ await checkOutputFilenames(outputFilenames, false, true, false)
687
+
688
+ const { detectedLanguage, detectedLanguageProbabilities } = await API.detectSpeechLanguage(inputFilePath, options)
689
+
690
+ results = detectedLanguageProbabilities
691
+ }
692
+
693
+ if (outputFilenames.length > 0) {
694
+ logger.start("\nWriting output files")
695
+
696
+ const resultsAsText = results.map(result => `${formatLanguageCodeWithName(result.language)}: ${result.probability.toFixed(5)}`).join("\n")
697
+
698
+ for (const outputFile of outputFilenames) {
699
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
700
+
701
+ await fileSaver(getEmptyRawAudio(0, 0), results as any, resultsAsText)
702
+ }
703
+ } else {
704
+ const resultsAsText = results.slice(0, 10).map(result => `${formatLanguageCodeWithName(result.language)}: ${result.probability.toFixed(5)}`).join("\n")
705
+
706
+ logger.log("")
707
+ logger.log(resultsAsText)
708
+ }
709
+
710
+ logger.end()
711
+ }
712
+
713
+ async function detectVoiceActivity(commandArgs: string[], cliOptions: Map<string, string>) {
714
+ const logger = new Logger()
715
+
716
+ const audioFilename = commandArgs[0]
717
+ const outputFilenames = commandArgs.slice(1)
718
+
719
+ if (audioFilename == undefined) {
720
+ throw new Error(`detect-voice-activity requires an argument containing the audio file path.`)
721
+ }
722
+
723
+ if (!existsSync(audioFilename)) {
724
+ throw new Error(`The given source audio file '${audioFilename}' was not found.`)
725
+ }
726
+
727
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
728
+ additionalOptionsSchema.set('play', { type: 'boolean' })
729
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
730
+
731
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
732
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
733
+ }
734
+
735
+ const options: API.VADOptions = await cliOptionsMapToOptionsObject(cliOptions, "VADOptions", additionalOptionsSchema)
736
+
737
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
738
+
739
+ await checkOutputFilenames(outputFilenames, true, true, false)
740
+
741
+ const { timeline, inputRawAudio } = await API.detectVoiceActivity(audioFilename, options)
742
+
743
+ if (outputFilenames.length > 0) {
744
+ logger.start("\nWriting output files")
745
+ }
746
+
747
+ for (const outputFile of outputFilenames) {
748
+ const partPatternMatch = outputFile.match(segmentFilenamePattern)
749
+
750
+ if (partPatternMatch) {
751
+ continue
752
+ }
753
+
754
+ const fileSaver = getFileSaver(outputFile, allowOverwrite)
755
+
756
+ await fileSaver(inputRawAudio, timeline, "")
757
+ }
758
+
759
+ logger.end()
760
+
761
+ if ((options as any).play) {
762
+ const normalizedAudio = normalizeAudioLevel(inputRawAudio)
763
+
764
+ const timelineToPlay = timeline.map(entry => {
765
+ return {...entry, type: "word" } as TimelineEntry
766
+ })
767
+
768
+ await playAudioWithWordTimeline(normalizedAudio, timelineToPlay)
769
+ }
770
+ }
771
+
772
+ async function denoise(commandArgs: string[], cliOptions: Map<string, string>) {
773
+ const logger = new Logger()
774
+
775
+ const audioFilename = commandArgs[0]
776
+ const outputFilenames = commandArgs.slice(1)
777
+
778
+ if (audioFilename == undefined) {
779
+ throw new Error(`'denoise' requires an argument containing the audio file path.`)
780
+ }
781
+
782
+ if (!existsSync(audioFilename)) {
783
+ throw new Error(`The given source audio file '${audioFilename}' was not found.`)
784
+ }
785
+
786
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
787
+ additionalOptionsSchema.set('play', { type: 'boolean' })
788
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
789
+
790
+ if (!cliOptions.has('play') && !cliOptions.has('no-play')) {
791
+ cliOptions.set('play', `${outputFilenames.length == 0}`)
792
+ }
793
+
794
+ const options: API.DenoisingOptions = await cliOptionsMapToOptionsObject(cliOptions, "DenoisingOptions", additionalOptionsSchema)
795
+
796
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
797
+
798
+ await checkOutputFilenames(outputFilenames, true, false, false)
799
+
800
+ const { denoisedAudio } = await API.denoise(audioFilename, options)
801
+
802
+ if (outputFilenames.length > 0) {
803
+ logger.start("\nWriting output files")
804
+ }
805
+
806
+ for (const filename of outputFilenames) {
807
+ const fileSaver = getFileSaver(filename, allowOverwrite)
808
+
809
+ await fileSaver(denoisedAudio, [], "")
810
+ }
811
+
812
+ logger.end()
813
+
814
+ if ((options as any).play) {
815
+ await playAudioSamples(denoisedAudio)
816
+ }
817
+ }
818
+
819
+ async function listEngines(commandArgs: string[], cliOptions: Map<string, string>) {
820
+ const logger = new Logger()
821
+
822
+ const targetOperation = commandArgs[0]
823
+
824
+ if (!targetOperation) {
825
+ throw new Error(`The 'list-engines' command requires an argument specifying the operation to list engines for, like 'echogarden list-engines transcribe'.`)
826
+ }
827
+
828
+ let engines: API.EngineMetadata[]
829
+
830
+ switch (targetOperation) {
831
+ case 'speak':
832
+ case 'speak-file':
833
+ case 'speak-url':
834
+ case 'speak-wikipedia': {
835
+ engines = API.synthesisEngines
836
+
837
+ break
838
+ }
839
+
840
+ case 'transcribe': {
841
+ engines = API.recognitionEngines
842
+
843
+ break
844
+ }
845
+
846
+ case 'align': {
847
+ engines = API.alignmentEngines
848
+
849
+ break
850
+ }
851
+
852
+ case 'translate-speech': {
853
+ engines = API.speechTranslationEngines
854
+
855
+ break
856
+ }
857
+
858
+ case 'detect-language': {
859
+ engines = [...API.speechLanguageDetectionEngines, ...API.textLanguageDetectionEngines]
860
+
861
+ break
862
+ }
863
+
864
+ case 'detect-speech-language': {
865
+ engines = API.speechLanguageDetectionEngines
866
+
867
+ break
868
+ }
869
+
870
+ case 'detect-text-language': {
871
+ engines = API.textLanguageDetectionEngines
872
+
873
+ break
874
+ }
875
+
876
+ case 'detect-voice-activity': {
877
+ engines = API.vadEngines
878
+
879
+ break
880
+ }
881
+
882
+ case 'denoise': {
883
+ engines = API.denoisingEngines
884
+
885
+ break
886
+ }
887
+
888
+ case 'list-voices': {
889
+ engines = API.synthesisEngines
890
+
891
+ break
892
+ }
893
+
894
+ case 'list-engines':
895
+ case 'install':
896
+ case 'uninstall':
897
+ case 'list-packages': {
898
+ throw new Error(`The operation '${targetOperation}' is not associated with a list of engines.`)
899
+ }
900
+
901
+ default: {
902
+ throw new Error(`Unrecognized operation: '${targetOperation}'`)
903
+ }
904
+ }
905
+
906
+ for (const [index, engine] of engines.entries()) {
907
+ logger.logTitledMessage('Identifier', chalk.magentaBright(engine.id))
908
+ logger.logTitledMessage('Name', engine.name)
909
+ logger.logTitledMessage('Description', engine.description)
910
+ logger.logTitledMessage('Type', engine.type)
911
+
912
+ if (index < engines.length - 1) {
913
+ logger.log("")
914
+ }
915
+ }
916
+ }
917
+
918
+ async function listTTSVoices(commandArgs: string[], cliOptions: Map<string, string>) {
919
+ const logger = new Logger()
920
+
921
+ const targetEngine = commandArgs[0]
922
+ const outputFilenames = commandArgs.slice(1)
923
+
924
+ if (!targetEngine) {
925
+ const optionsSchema = await getOptionsSchema()
926
+ const { enum: ttsEnginesEnum } = getOptionTypeFromSchema(["VoiceListRequestOptions", "engine"], optionsSchema)
927
+
928
+ throw new Error(`list-voices requires an argument specifying one of these supported engines:\n${ttsEnginesEnum!.join(", ")}`)
929
+ }
930
+
931
+ const additionalOptionsSchema = new Map<string, SchemaTypeDefinition>()
932
+ additionalOptionsSchema.set('overwrite', { type: 'boolean' })
933
+
934
+ cliOptions.set('engine', targetEngine)
935
+
936
+ const options: API.VoiceListRequestOptions = await cliOptionsMapToOptionsObject(cliOptions, "VoiceListRequestOptions", additionalOptionsSchema)
937
+
938
+ const allowOverwrite = getWithDefault((options as any).overwrite, overwriteByDefault)
939
+
940
+ await checkOutputFilenames(outputFilenames, false, true, false)
941
+
942
+ const { voiceList } = await API.requestVoiceList(options)
943
+
944
+ const voiceListText = voiceList.map(entry => {
945
+ const nameText = entry.name
946
+ const languagesNamesText = entry.languages.map(language => formatLanguageCodeWithName(language)).join(", ")
947
+ const genderText = entry.gender
948
+
949
+ let entryText = `${chalk.cyanBright('Identifier')}: ${chalk.magentaBright(nameText)}\n${chalk.cyanBright('Languages')}: ${languagesNamesText}\n${chalk.cyanBright('Gender')}: ${genderText}`
950
+
951
+ const speakerCount = entry.speakerCount
952
+
953
+ if (speakerCount) {
954
+ entryText += `\n${chalk.cyanBright('Speaker count')}: ${speakerCount}`
955
+ }
956
+
957
+ return entryText
958
+ }).join("\n\n")
959
+
960
+ if (outputFilenames.length > 0) {
961
+ logger.start("\nWriting output files")
962
+
963
+ for (const filename of outputFilenames) {
964
+ const fileSaver = getFileSaver(filename, allowOverwrite)
965
+
966
+ const { default: stripAnsi } = await import('strip-ansi')
967
+ const voiceListTextWithoutColors = stripAnsi(voiceListText)
968
+
969
+ await fileSaver(getEmptyRawAudio(0, 0), voiceList as any, voiceListTextWithoutColors)
970
+ }
971
+ } else {
972
+ logger.log(voiceListText)
973
+ }
974
+
975
+ logger.end()
976
+ }
977
+
978
+ async function installPackages(commandArgs: string[], cliOptions: Map<string, string>) {
979
+ const logger = new Logger()
980
+
981
+ if (commandArgs.length == 0) {
982
+ throw new Error("No package names specified")
983
+ }
984
+
985
+ const failedPackageNames: string[] = []
986
+
987
+ for (const packageName of commandArgs) {
988
+ try {
989
+ await loadPackage(packageName)
990
+ } catch (e) {
991
+ resetActiveLogger()
992
+
993
+ logger.log(`Failed installing package ${packageName}: ${e}`)
994
+ failedPackageNames.push(packageName)
995
+ }
996
+ }
997
+
998
+ if (failedPackageNames.length > 0) {
999
+ if (failedPackageNames.length == 1) {
1000
+ logger.log(`The package ${failedPackageNames[0]} failed to install`)
1001
+ } else {
1002
+ logger.log(`The packages ${failedPackageNames.join(', ')} failed to install`)
1003
+ }
1004
+ }
1005
+ }
1006
+
1007
+ async function uninstallPackages(commandArgs: string[], cliOptions: Map<string, string>) {
1008
+ const logger = new Logger()
1009
+
1010
+ if (commandArgs.length == 0) {
1011
+ throw new Error("No package names specified")
1012
+ }
1013
+
1014
+ const failedPackageNames: string[] = []
1015
+
1016
+ for (const packageName of commandArgs) {
1017
+ try {
1018
+ await removePackage(packageName)
1019
+ } catch (e) {
1020
+ resetActiveLogger()
1021
+
1022
+ logger.log(`Failed uninstalling package ${packageName}: ${e}`)
1023
+ failedPackageNames.push(packageName)
1024
+ }
1025
+ }
1026
+
1027
+ if (failedPackageNames.length > 0) {
1028
+ logger.log(`The packages ${failedPackageNames.join(', ')} failed to uninstall`)
1029
+ }
1030
+ }
1031
+
1032
+ async function listPackages(commandArgs: string[], cliOptions: Map<string, string>) {
1033
+ const logger = new Logger()
1034
+
1035
+ const packagesDir = await ensureAndGetPackagesDir()
1036
+
1037
+ const installedPackageNames = await readdir(packagesDir)
1038
+
1039
+ const installedPackageNamesFormatted = installedPackageNames.map(packageName => {
1040
+ const versionTag = getVersionTagFromPackageName(packageName)
1041
+
1042
+ let unversionedPackageName = packageName
1043
+
1044
+ if (versionTag) {
1045
+ unversionedPackageName = packageName.substring(0, packageName.length - versionTag.length - 1)
1046
+ }
1047
+
1048
+ const resolvedVersionTag = resolveVersionTagForUnversionedPackageName(unversionedPackageName)
1049
+
1050
+ if (resolvedVersionTag == versionTag) {
1051
+ return packageName
1052
+ } else {
1053
+ return `${packageName} (unused)`
1054
+ }
1055
+ })
1056
+
1057
+ installedPackageNamesFormatted.sort()
1058
+
1059
+ logger.log(`Total of ${installedPackageNamesFormatted.length} packages installed in '${packagesDir}'\n`)
1060
+
1061
+ logger.log(installedPackageNamesFormatted.join("\n"))
1062
+ }
1063
+
1064
+ async function serve(commandArgs: string[], cliOptions: Map<string, string>) {
1065
+ const options: ServerOptions = await cliOptionsMapToOptionsObject(cliOptions, "ServerOptions")
1066
+
1067
+ async function onServerStarted(serverOptions: ServerOptions) {
1068
+ // Run a test routine (early development)
1069
+ //await runClientWebSocketTest(serverOptions.port!, serverOptions.secure!)
1070
+ }
1071
+
1072
+ await startServer(options, onServerStarted)
1073
+ }
1074
+
1075
+ async function cliOptionsMapToOptionsObject(cliOptionsMap: Map<string, string>, optionsRoot: keyof APIOptions, additionalOptionsSchema?: Map<string, SchemaTypeDefinition>) {
1076
+ const optionsSchema = await getOptionsSchema()
1077
+ const resultingObj: any = {}
1078
+
1079
+ function setValueAtPath(path: string[], value: any) {
1080
+ let currentObject = resultingObj
1081
+
1082
+ for (let keyIndex = 0; keyIndex < path.length; keyIndex++) {
1083
+ const key = path[keyIndex]
1084
+
1085
+ if (keyIndex == path.length - 1) {
1086
+ currentObject[key] = value
1087
+ } else {
1088
+ if (!(key in currentObject)) {
1089
+ currentObject[key] = {}
1090
+ }
1091
+
1092
+ currentObject = currentObject[key]
1093
+ }
1094
+ }
1095
+ }
1096
+
1097
+ for (let [key, value] of cliOptionsMap) {
1098
+ let isNegated = false
1099
+
1100
+ if (key.startsWith("no-")) {
1101
+ isNegated = true
1102
+ key = key.slice(3)
1103
+
1104
+ if (value) {
1105
+ throw new Error(`The negated property '${key}' cannot have a value.`)
1106
+ }
1107
+ }
1108
+
1109
+ const path = key.split(".")
1110
+
1111
+ let optionType: string | undefined
1112
+ let optionEnum: any[] | undefined
1113
+ let optionIsUnion: boolean | undefined
1114
+
1115
+ if (additionalOptionsSchema && additionalOptionsSchema.has(key)) {
1116
+ ({ type: optionType, enum: optionEnum, isUnion: optionIsUnion } = additionalOptionsSchema.get(key)!)
1117
+ } else {
1118
+ const extendedPath = [optionsRoot, ...path];
1119
+ ({ type: optionType, enum: optionEnum, isUnion: optionIsUnion } = getOptionTypeFromSchema(extendedPath, optionsSchema))
1120
+ }
1121
+
1122
+ let parsedValue: any
1123
+
1124
+ if (optionType == 'string') {
1125
+ parsedValue = value
1126
+ } else if (optionType == 'number') {
1127
+ parsedValue = parseFloat(value)
1128
+
1129
+ if (isNaN(parsedValue)) {
1130
+ throw new Error(`The property '${key}' is a number. '${value}' cannot be parsed as a number.`)
1131
+ }
1132
+ } else if (optionType == 'boolean') {
1133
+ if (value == null || value == '') {
1134
+ parsedValue = !isNegated
1135
+ } else if (value == 'true') {
1136
+ parsedValue = true
1137
+ } else if (value == 'false') {
1138
+ parsedValue = false
1139
+ } else {
1140
+ throw new Error(`The property '${key}' is a Boolean, which can receive either 'true' or 'false', not '${value}'.`)
1141
+ }
1142
+ } else if (value == null || value == '') {
1143
+ throw new Error(`No value was specified for the property '${key}', which has type ${optionType}.`)
1144
+ } else if (isNegated) {
1145
+ throw new Error(`The property '${key}' is not a Boolean, and cannot be negated using the 'not-' prefix.`)
1146
+ } else if (optionType == 'array' || optionType == 'object') {
1147
+ try {
1148
+ const { default: JSON5 } = await import('json5')
1149
+ parsedValue = JSON5.parse(value)
1150
+ } catch (e) {
1151
+ parsedValue = value
1152
+ }
1153
+ } else if (optionIsUnion) {
1154
+ const isArrayJSON = value.startsWith('[') && value.endsWith(']')
1155
+ const isObjectJSON = value.startsWith('{') && value.endsWith('}')
1156
+ const isNumberJSON = !isNaN(Number.parseFloat(value))
1157
+ const isBooleanJSON = value == 'true' || value == 'false'
1158
+
1159
+ if (isArrayJSON || isObjectJSON || isNumberJSON || isBooleanJSON) {
1160
+ parsedValue = JSON5.parse(value)
1161
+ } else {
1162
+ parsedValue = value
1163
+ }
1164
+ } else {
1165
+ parsedValue = value
1166
+ }
1167
+
1168
+ if (optionEnum && !optionEnum.includes(parsedValue)) {
1169
+ throw new Error(`The property '${key}' must be one of ${optionEnum.join(", ")}`)
1170
+ }
1171
+
1172
+ setValueAtPath(path, parsedValue)
1173
+ }
1174
+
1175
+ return resultingObj
1176
+ }
1177
+
1178
+ let cachedOptionsSchema: any
1179
+ export async function getOptionsSchema() {
1180
+ if (!cachedOptionsSchema) {
1181
+ cachedOptionsSchema = await readAndParseJsonFile(resolveToModuleRootDir("data/schemas/options.json"))
1182
+ }
1183
+
1184
+ return cachedOptionsSchema
1185
+ }
1186
+
1187
+ export async function checkOutputFilenames(outputFilenames: string[], acceptMediaOutputs: boolean, acceptMetadataOutputs: boolean, acceptSubtitleOutputs: boolean) {
1188
+ const supportedFileExtensions: string[] = []
1189
+
1190
+ if (acceptMediaOutputs) {
1191
+ supportedFileExtensions.push(...supportedOutputMediaFileExtensions)
1192
+ }
1193
+
1194
+ if (acceptMetadataOutputs) {
1195
+ supportedFileExtensions.push(...supportedMetadataFileExtensions)
1196
+ }
1197
+
1198
+ if (acceptSubtitleOutputs) {
1199
+ supportedFileExtensions.push(...supportedSubtitleFileExtensions)
1200
+ }
1201
+
1202
+ let includesPartPattern = false
1203
+
1204
+ for (const outputFilename of outputFilenames) {
1205
+ const fileExtension = getLowercaseFileExtension(outputFilename)
1206
+
1207
+ if (!supportedFileExtensions.includes(fileExtension)) {
1208
+ let errorText = ""
1209
+ errorText += `\nThe specified output path '${outputFilename}' doesn't have a supported file extension.\n`
1210
+ errorText += `\nSupported extensions are:\n`
1211
+
1212
+ if (acceptMediaOutputs) {
1213
+ errorText += `${formatListWithQuotedElements(supportedOutputMediaFileExtensions)} for audio output files.\n`
1214
+ }
1215
+
1216
+ if (acceptMetadataOutputs) {
1217
+ errorText += `${formatListWithQuotedElements(supportedMetadataFileExtensions)} for metadata output files.\n`
1218
+ }
1219
+
1220
+ if (acceptMetadataOutputs) {
1221
+ errorText += `${formatListWithQuotedElements(supportedSubtitleFileExtensions)} for subtitle output files.\n`
1222
+ }
1223
+
1224
+ throw new Error(errorText)
1225
+ }
1226
+
1227
+ const partPatternMatch = outputFilename.match(segmentFilenamePattern)
1228
+
1229
+ if (partPatternMatch) {
1230
+ if (partPatternMatch[1] != "segment") {
1231
+ throw new Error(`Invalid square bracket pattern: '${partPatternMatch[1]}'. Square bracket output filename pattern currently only supports the value 'segment'. For example: '/segment/[segment].wav'`)
1232
+ }
1233
+
1234
+ includesPartPattern = true
1235
+ }
1236
+ }
1237
+
1238
+ return { includesPartPattern }
1239
+ }
1240
+
1241
+ export async function writeOutputFilesForSegment(outputFilenames: string[], index: number, total: number, audio: RawAudio, timeline: Timeline, text: string, language: string, allowOverwrite: boolean) {
1242
+ const digitCount = Math.max((total + 1).toString().length, 2)
1243
+
1244
+ const segmentWords = (await splitToWords(text, language)).filter(text => wordCharacterPattern.test(text))
1245
+
1246
+ const segmentJoinedWords = segmentWords.join(" ").trim()
1247
+
1248
+ let initialText: string
1249
+
1250
+ const maxLength = 50
1251
+
1252
+ if (segmentJoinedWords.length < maxLength) {
1253
+ initialText = segmentJoinedWords.substring(0, maxLength).trim()
1254
+ } else {
1255
+ initialText = segmentJoinedWords.substring(0, maxLength - 4).trim() + ".. "
1256
+ }
1257
+
1258
+ for (const outputFilename of outputFilenames) {
1259
+ const partPatternMatch = outputFilename.match(segmentFilenamePattern)
1260
+
1261
+ if (!partPatternMatch) {
1262
+ continue
1263
+ }
1264
+
1265
+ const segmentFilename = outputFilename.replace(segmentFilenamePattern, `${formatIntegerWithLeadingZeros(index + 1, digitCount)} - ${initialText}.$2`)
1266
+
1267
+ const fileSaver = getFileSaver(segmentFilename, allowOverwrite)
1268
+ await fileSaver(audio, timeline, text)
1269
+ }
1270
+ }
1271
+
1272
+ type FileSaver = (audio: RawAudio, timeline: Timeline, text: string, subtitlesConfig?: SubtitlesConfig) => Promise<void>
1273
+
1274
+ function getFileSaver(outputFilePath: string, allowOverwrite: boolean): FileSaver {
1275
+ const parsedPath = parsePath(outputFilePath)
1276
+
1277
+ const fileDir = parsedPath.dir || "./"
1278
+
1279
+ if (!allowOverwrite) {
1280
+ const filenameParts = splitFilenameOnExtendedExtension(parsedPath.base)
1281
+
1282
+ for (let i = 1; existsSync(outputFilePath); i++) {
1283
+ outputFilePath = path.join(parsedPath.dir, `${filenameParts[0]} (${i})`)
1284
+
1285
+ if (filenameParts[1]) {
1286
+ outputFilePath += `.${filenameParts[1]}`
1287
+ }
1288
+ }
1289
+ }
1290
+
1291
+ const fileExtension = getLowercaseFileExtension(outputFilePath)
1292
+
1293
+ let fileSaver: FileSaver
1294
+
1295
+ if (fileExtension == "txt") {
1296
+ fileSaver = async (audio, timeline, text) => {
1297
+ await ensureDir(fileDir)
1298
+ return writeFileSafe(outputFilePath, text, { encoding: "utf-8" })
1299
+ }
1300
+ } else if (fileExtension == "json") {
1301
+ fileSaver = async (audio, timeline, text) => {
1302
+ await ensureDir(fileDir)
1303
+
1304
+ const roundedTimeline = roundTimelineProperties(timeline)
1305
+ return writeFileSafe(outputFilePath, stringifyAndFormatJson(roundedTimeline))
1306
+ }
1307
+ } else if (fileExtension == "srt") {
1308
+ fileSaver = async (audio, timeline, text, subtitlesConfig) => {
1309
+ await ensureDir(fileDir)
1310
+
1311
+ subtitlesConfig = extendDeep(subtitlesConfig, { format: "srt" })
1312
+
1313
+ const subtitles = timelineToSubtitles(timeline, subtitlesConfig)
1314
+
1315
+ return writeFileSafe(outputFilePath, subtitles)
1316
+ }
1317
+ } else if (fileExtension == "vtt") {
1318
+ fileSaver = async (audio, timeline, text, subtitlesConfig) => {
1319
+ await ensureDir(fileDir)
1320
+
1321
+ subtitlesConfig = extendDeep(subtitlesConfig, { format: "webvtt" })
1322
+
1323
+ const subtitles = timelineToSubtitles(timeline, subtitlesConfig)
1324
+
1325
+ return writeFileSafe(outputFilePath, subtitles)
1326
+ }
1327
+ } else if (fileExtension == "wav") {
1328
+ fileSaver = async (audio) => {
1329
+ await ensureDir(fileDir)
1330
+
1331
+ return writeFileSafe(outputFilePath, encodeWaveBuffer(audio))
1332
+ }
1333
+ } else if (supportedOutputMediaFileExtensions.includes(fileExtension)) {
1334
+ fileSaver = async (audio) => {
1335
+ const ffmpegOptions = getDefaultFFMpegOptionsForSpeech(fileExtension)
1336
+
1337
+ ffmpegOptions.filename = outputFilePath
1338
+
1339
+ await ensureDir(fileDir)
1340
+
1341
+ await encodeFromChannels(audio, ffmpegOptions)
1342
+
1343
+ return
1344
+ }
1345
+ } else {
1346
+ throw new Error("Unsupported file extension")
1347
+ }
1348
+
1349
+ return fileSaver
1350
+ }
1351
+
1352
+ const supportedMetadataFileExtensions = ['txt', 'json']
1353
+ const supportedSubtitleFileExtensions = ['srt', 'vtt']
1354
+ const supportedOutputMediaFileExtensions = ["wav", "mp3", "opus", "m4a", "ogg", "flac"]
1355
+
1356
+ const segmentFilenamePattern = /\[(.*)\]\.(.*)$/
1357
+
1358
+ const overwriteByDefault = false
1359
+
1360
+ startIfInWorkerThread()