echogarden 0.11.12 → 0.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +16 -0
- package/dist/api/Alignment.js +2 -2
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +2 -2
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +5 -4
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +2 -2
- package/dist/api/Translation.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +1 -0
- package/dist/audio/AudioUtilities.js +25 -7
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.js +2 -2
- package/dist/cli/CLI.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +2 -2
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/subtitles/Subtitles.d.ts +10 -7
- package/dist/subtitles/Subtitles.js +268 -207
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/docs/Options.md +4 -2
- package/package.json +7 -6
- package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
- package/src/alignment/DTWSequenceAlignment.ts +121 -0
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
- package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
- package/src/alignment/SpeechAlignment.ts +488 -0
- package/src/api/API.ts +12 -0
- package/src/api/APIOptions.ts +15 -0
- package/src/api/Alignment.ts +329 -0
- package/src/api/Common.ts +16 -0
- package/src/api/Denoising.ts +120 -0
- package/src/api/LanguageDetection.ts +286 -0
- package/src/api/Recognition.ts +344 -0
- package/src/api/Synthesis.ts +1735 -0
- package/src/api/Translation.ts +143 -0
- package/src/api/Vad.ts +172 -0
- package/src/audio/AudioBufferConversion.ts +248 -0
- package/src/audio/AudioPlayer.ts +358 -0
- package/src/audio/AudioRecorder.ts +91 -0
- package/src/audio/AudioUtilities.ts +392 -0
- package/src/audio/SoxPath.ts +24 -0
- package/src/cli/CLI.ts +1360 -0
- package/src/cli/CLIConfigFile.ts +91 -0
- package/src/cli/CLILauncher.ts +26 -0
- package/src/cli/CLIOptionsSchema.ts +54 -0
- package/src/cli/CLIParser.ts +41 -0
- package/src/cli/CLIStarter.ts +40 -0
- package/src/codecs/FFMpegTranscoder.ts +214 -0
- package/src/codecs/TIMITCodec.ts +17 -0
- package/src/codecs/WaveCodec.ts +260 -0
- package/src/denoising/RNNoise.ts +95 -0
- package/src/dsp/BiquadFilter.ts +488 -0
- package/src/dsp/FFT.ts +187 -0
- package/src/dsp/MFCC.ts +227 -0
- package/src/dsp/MelSpectogram.ts +145 -0
- package/src/dsp/Rubberband.ts +249 -0
- package/src/dsp/Sonic.ts +59 -0
- package/src/dsp/SpeexResampler.ts +79 -0
- package/src/math/VectorMath.ts +812 -0
- package/src/nlp/ChineseSegmentation.ts +68 -0
- package/src/nlp/CompromiseNLP.ts +113 -0
- package/src/nlp/EspeakPhonemizer.ts +168 -0
- package/src/nlp/IPA.ts +139 -0
- package/src/nlp/JapaneseSegmentation.ts +53 -0
- package/src/nlp/Lexicon.ts +119 -0
- package/src/nlp/PhoneConversion.ts +508 -0
- package/src/nlp/Segmentation.ts +237 -0
- package/src/nlp/TextNormalizer.ts +160 -0
- package/src/recognition/AmazonTranscribeSTT.ts +112 -0
- package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
- package/src/recognition/GoogleCloudSTT.ts +92 -0
- package/src/recognition/SileroSTT.ts +173 -0
- package/src/recognition/VoskSTT.ts +112 -0
- package/src/recognition/WhisperSTT.ts +1518 -0
- package/src/server/Client.ts +297 -0
- package/src/server/Server.ts +178 -0
- package/src/server/ServerStarter.ts +12 -0
- package/src/server/Worker.ts +400 -0
- package/src/server/WorkerStarter.ts +38 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
- package/src/subtitles/Subtitles.ts +478 -0
- package/src/synthesis/AwsPollyTTS.ts +78 -0
- package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
- package/src/synthesis/CoquiServerTTS.ts +29 -0
- package/src/synthesis/ElevenLabsTTS.ts +104 -0
- package/src/synthesis/EspeakTTS.ts +552 -0
- package/src/synthesis/FliteTTS.ts +387 -0
- package/src/synthesis/GoogleCloudTTS.ts +112 -0
- package/src/synthesis/GoogleTranslateTTS.ts +210 -0
- package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
- package/src/synthesis/SamTTS.ts +30 -0
- package/src/synthesis/SapiTTS.ts +222 -0
- package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
- package/src/synthesis/SvoxPicoTTS.ts +318 -0
- package/src/synthesis/VitsTTS.ts +734 -0
- package/src/tests/Test.ts +24 -0
- package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
- package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
- package/src/typings/Fillers.d.ts +41 -0
- package/src/utilities/BinaryArrayConversion.ts +159 -0
- package/src/utilities/Compression.ts +91 -0
- package/src/utilities/FileDownloader.ts +201 -0
- package/src/utilities/FileSystem.ts +265 -0
- package/src/utilities/Hashing.ts +230 -0
- package/src/utilities/Locale.ts +119 -0
- package/src/utilities/Logger.ts +72 -0
- package/src/utilities/NdArrayUtilities.ts +31 -0
- package/src/utilities/ObjectUtilities.ts +169 -0
- package/src/utilities/OpenPromise.ts +13 -0
- package/src/utilities/PackageManager.ts +97 -0
- package/src/utilities/Queue.ts +17 -0
- package/src/utilities/RandomGenerator.ts +237 -0
- package/src/utilities/SignalChannel.ts +22 -0
- package/src/utilities/TarballMaker.ts +68 -0
- package/src/utilities/Timeline.ts +231 -0
- package/src/utilities/Timer.ts +93 -0
- package/src/utilities/Utilities.ts +574 -0
- package/src/utilities/WasmMemoryManager.ts +516 -0
- package/src/utilities/WebReader.ts +55 -0
- package/src/utilities/WikipediaReader.ts +41 -0
- package/src/voice-activity-detection/SileroVAD.ts +86 -0
- package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
|
@@ -0,0 +1,358 @@
|
|
|
1
|
+
import { parentPort } from 'node:worker_threads'
|
|
2
|
+
|
|
3
|
+
import { ChildProcessWithoutNullStreams, spawn } from 'child_process'
|
|
4
|
+
|
|
5
|
+
import { RawAudio, encodeWaveBuffer, fadeAudioInOut, getRawAudioDuration, sliceRawAudioByTime } from "./AudioUtilities.js"
|
|
6
|
+
import * as AudioBufferConversion from './AudioBufferConversion.js'
|
|
7
|
+
import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
|
|
8
|
+
|
|
9
|
+
import { Timer } from "../utilities/Timer.js"
|
|
10
|
+
import { getRandomHexString, waitTimeout, writeToStderr } from '../utilities/Utilities.js'
|
|
11
|
+
import { encodeToAudioBuffer } from './AudioBufferConversion.js'
|
|
12
|
+
import { OpenPromise } from '../utilities/OpenPromise.js'
|
|
13
|
+
import { Timeline, addWordTextOffsetsToTimeline } from '../utilities/Timeline.js'
|
|
14
|
+
import { getAppTempDir, outputFile, readAndParseJsonFile, readFile, remove } from '../utilities/FileSystem.js'
|
|
15
|
+
import { tryResolvingSoxPath } from './SoxPath.js'
|
|
16
|
+
import { SignalChannel } from '../utilities/SignalChannel.js'
|
|
17
|
+
import { deepClone } from '../utilities/ObjectUtilities.js'
|
|
18
|
+
import { appName } from '../api/Common.js'
|
|
19
|
+
import path from 'node:path'
|
|
20
|
+
|
|
21
|
+
export async function playAudioFileWithTimelineFile(audioFilename: string, timelineFileName: string, transcriptFileName?: string) {
|
|
22
|
+
const rawAudio = await FFMpegTranscoder.decodeToChannels(audioFilename, 48000, 1)
|
|
23
|
+
|
|
24
|
+
const timeline = await readAndParseJsonFile(timelineFileName)
|
|
25
|
+
|
|
26
|
+
let transcript: string | undefined
|
|
27
|
+
if (transcriptFileName) {
|
|
28
|
+
transcript = await readFile(transcriptFileName, "utf8")
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
await playAudioWithWordTimeline(rawAudio, timeline, transcript)
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export async function playAudioWithWordTimeline(rawAudio: RawAudio, wordTimeline: Timeline, transcript?: string) {
|
|
35
|
+
if (!transcript) {
|
|
36
|
+
transcript = wordTimeline.map(entry => entry.text).join(' ')
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
wordTimeline = deepClone(wordTimeline)
|
|
40
|
+
|
|
41
|
+
addWordTextOffsetsToTimeline(wordTimeline, transcript)
|
|
42
|
+
|
|
43
|
+
let timelineEntryIndex = 0
|
|
44
|
+
let transcriptOffset = 0
|
|
45
|
+
|
|
46
|
+
function onTimePosition(timePosition: number) {
|
|
47
|
+
const text = transcript!
|
|
48
|
+
|
|
49
|
+
for (; timelineEntryIndex < wordTimeline.length; timelineEntryIndex++) {
|
|
50
|
+
const entry = wordTimeline[timelineEntryIndex]
|
|
51
|
+
|
|
52
|
+
if (entry.startTime > timePosition) {
|
|
53
|
+
return
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const wordStartOffset = entry.startOffsetUtf16
|
|
57
|
+
let wordEndOffset = entry.endOffsetUtf16
|
|
58
|
+
|
|
59
|
+
if (wordStartOffset == null || wordEndOffset == null) {
|
|
60
|
+
//writeToStderr(` [No offset availble for '${entry.text}'] `)
|
|
61
|
+
|
|
62
|
+
continue
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
while (wordEndOffset < text.length &&
|
|
66
|
+
charactersToWriteAhead.includes(text[wordEndOffset]) &&
|
|
67
|
+
text[wordEndOffset] != wordTimeline[timelineEntryIndex + 2]?.text) {
|
|
68
|
+
wordEndOffset += 1
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
writeToStderr(text.substring(transcriptOffset, wordEndOffset))
|
|
72
|
+
|
|
73
|
+
transcriptOffset = wordEndOffset
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
writeToStderr("\n")
|
|
78
|
+
|
|
79
|
+
const signalChannel = new SignalChannel()
|
|
80
|
+
|
|
81
|
+
const keypressListenerStartTimestamp = Date.now()
|
|
82
|
+
|
|
83
|
+
function keypressHandler(message: any) {
|
|
84
|
+
if (message.name == "keypress" &&
|
|
85
|
+
message.key.name == 'return' &&
|
|
86
|
+
message.timestamp >= keypressListenerStartTimestamp) {
|
|
87
|
+
|
|
88
|
+
signalChannel.send("abort")
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
parentPort?.on('message', keypressHandler)
|
|
93
|
+
|
|
94
|
+
await playAudioSamples(rawAudio, onTimePosition, signalChannel)
|
|
95
|
+
|
|
96
|
+
parentPort?.off('message', keypressHandler)
|
|
97
|
+
|
|
98
|
+
writeToStderr("\n")
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export async function playAudioWithTimelinePhones(rawAudio: RawAudio, timeline: Timeline) {
|
|
102
|
+
let wordIndex = 0
|
|
103
|
+
let phoneIndex = 0
|
|
104
|
+
|
|
105
|
+
function onTimePosition(timePosition: number) {
|
|
106
|
+
for (; wordIndex < timeline.length; wordIndex++) {
|
|
107
|
+
const wordEntry = timeline[wordIndex]
|
|
108
|
+
const phoneTimeline = wordEntry.timeline!
|
|
109
|
+
|
|
110
|
+
if (phoneTimeline.every(phoneEntry => phoneEntry.text == "")) {
|
|
111
|
+
continue
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
for (; phoneIndex < phoneTimeline.length; phoneIndex++) {
|
|
115
|
+
const entry = phoneTimeline[phoneIndex]
|
|
116
|
+
|
|
117
|
+
if (entry.startTime > timePosition) {
|
|
118
|
+
return
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
writeToStderr(`${entry.text} `)
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
writeToStderr(`| `)
|
|
125
|
+
phoneIndex = 0
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
writeToStderr("\n")
|
|
130
|
+
|
|
131
|
+
await playAudioSamples(rawAudio, onTimePosition)
|
|
132
|
+
|
|
133
|
+
writeToStderr("\n")
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
export async function playAudioPairWithTimelineInterleaved(rawAudio1: RawAudio, rawAudio2: RawAudio, timeline1: Timeline, timeline2: Timeline) {
|
|
138
|
+
if (timeline1.length != timeline2.length) {
|
|
139
|
+
throw new Error("Timelines have different lengths")
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
for (let i = 0; i < timeline1.length; i++) {
|
|
143
|
+
writeToStderr(`${timeline1[i].text} `)
|
|
144
|
+
|
|
145
|
+
await playAudioSamples(sliceRawAudioByTime(rawAudio1, timeline1[i].startTime, timeline1[i].endTime))
|
|
146
|
+
await waitTimeout(200)
|
|
147
|
+
await playAudioSamples(sliceRawAudioByTime(rawAudio2, timeline2[i].startTime, timeline2[i].endTime))
|
|
148
|
+
await waitTimeout(500)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
writeToStderr("\n")
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
export function playAudioSamples(rawAudio: RawAudio, onTimePosition?: (timePosition: number) => void, signalChannel?: SignalChannel, microFadeInOut = true) {
|
|
155
|
+
return new Promise<void>(async (resolve, reject) => {
|
|
156
|
+
if (microFadeInOut) {
|
|
157
|
+
rawAudio = fadeAudioInOut(rawAudio, 0.0025)
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
let playerProcessClosed = false
|
|
161
|
+
|
|
162
|
+
const channelCount = rawAudio.audioChannels.length
|
|
163
|
+
const audioDuration = getRawAudioDuration(rawAudio)
|
|
164
|
+
|
|
165
|
+
const playerSpawnedOpenPromise = new OpenPromise<null>()
|
|
166
|
+
|
|
167
|
+
const soxPath = await tryResolvingSoxPath()
|
|
168
|
+
|
|
169
|
+
let aborted = false
|
|
170
|
+
|
|
171
|
+
if (!soxPath) {
|
|
172
|
+
throw new Error(`Couldn't find or install the SoX utility. Please install the SoX utility on your system path to enable audio playback.`)
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
let streamToStdin = true
|
|
176
|
+
|
|
177
|
+
if (process.platform == "darwin") {
|
|
178
|
+
streamToStdin = false
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
let tempFilePath: string | undefined
|
|
182
|
+
let audioBuffer: Buffer | undefined
|
|
183
|
+
|
|
184
|
+
async function cleanup() {
|
|
185
|
+
if (tempFilePath) {
|
|
186
|
+
await remove(tempFilePath)
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
let playerProcess: ChildProcessWithoutNullStreams
|
|
191
|
+
|
|
192
|
+
if (streamToStdin) {
|
|
193
|
+
audioBuffer = encodeToAudioBuffer(rawAudio.audioChannels)
|
|
194
|
+
|
|
195
|
+
playerProcess = spawn(
|
|
196
|
+
soxPath,
|
|
197
|
+
['-t', 'raw', '-r', `${rawAudio.sampleRate}`, '-e', 'signed', '-b', '16', '-c', channelCount.toString(), '-', '-d'],
|
|
198
|
+
{}
|
|
199
|
+
)
|
|
200
|
+
} else {
|
|
201
|
+
tempFilePath = path.join(getAppTempDir(appName), `${getRandomHexString(16)}.wav`)
|
|
202
|
+
const waveFileBuffer = encodeWaveBuffer(rawAudio)
|
|
203
|
+
await outputFile(tempFilePath, waveFileBuffer)
|
|
204
|
+
|
|
205
|
+
playerProcess = spawn(
|
|
206
|
+
soxPath,
|
|
207
|
+
[tempFilePath, '-d'],
|
|
208
|
+
{}
|
|
209
|
+
)
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
if (signalChannel) {
|
|
213
|
+
signalChannel.on("abort", () => {
|
|
214
|
+
aborted = true
|
|
215
|
+
playerProcess.kill('SIGKILL')
|
|
216
|
+
})
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
// Required to work around SoX bug:
|
|
220
|
+
playerProcess.stderr.on("data", (data) => {
|
|
221
|
+
//writeToStderr(data.toString('utf-8'))
|
|
222
|
+
})
|
|
223
|
+
|
|
224
|
+
playerProcess.stdout.on("data", (data) => {
|
|
225
|
+
//writeToStderr(data.toString('utf-8'))
|
|
226
|
+
})
|
|
227
|
+
|
|
228
|
+
playerProcess.once("spawn", () => {
|
|
229
|
+
if (audioBuffer != undefined) {
|
|
230
|
+
playerProcess.stdin!.write(audioBuffer)
|
|
231
|
+
playerProcess.stdin!.end()
|
|
232
|
+
playerProcess.stdin!.on("error", () => { })
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
playerSpawnedOpenPromise.resolve(null)
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
playerProcess.once("error", async (e) => {
|
|
239
|
+
await cleanup()
|
|
240
|
+
playerProcessClosed = true
|
|
241
|
+
|
|
242
|
+
reject(e)
|
|
243
|
+
})
|
|
244
|
+
|
|
245
|
+
playerProcess.once('close', async () => {
|
|
246
|
+
await cleanup()
|
|
247
|
+
playerProcessClosed = true
|
|
248
|
+
|
|
249
|
+
resolve()
|
|
250
|
+
})
|
|
251
|
+
|
|
252
|
+
await playerSpawnedOpenPromise.promise
|
|
253
|
+
|
|
254
|
+
const timer = new Timer()
|
|
255
|
+
|
|
256
|
+
while (!playerProcessClosed && !aborted) {
|
|
257
|
+
const elapsedTime = timer.elapsedTimeSeconds
|
|
258
|
+
|
|
259
|
+
if (onTimePosition) {
|
|
260
|
+
onTimePosition(elapsedTime)
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
if (playerProcessClosed || elapsedTime >= audioDuration) {
|
|
264
|
+
if (onTimePosition) {
|
|
265
|
+
onTimePosition(audioDuration)
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
return
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
await waitTimeout(20)
|
|
272
|
+
}
|
|
273
|
+
})
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
export function playAudioSamples_Speaker(rawAudio: RawAudio, onTimePosition?: (timePosition: number) => void, microFadeInOut = true) {
|
|
277
|
+
return new Promise<void>(async (resolve, reject) => {
|
|
278
|
+
if (microFadeInOut) {
|
|
279
|
+
rawAudio = fadeAudioInOut(rawAudio, 0.0025)
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
const channelCount = rawAudio.audioChannels.length
|
|
283
|
+
let audioData = AudioBufferConversion.encodeToAudioBuffer(rawAudio.audioChannels)
|
|
284
|
+
|
|
285
|
+
const { default: Speaker } = await import('speaker')
|
|
286
|
+
|
|
287
|
+
const speaker = new Speaker({
|
|
288
|
+
channels: rawAudio.audioChannels.length,
|
|
289
|
+
bitDepth: 16,
|
|
290
|
+
sampleRate: rawAudio.sampleRate,
|
|
291
|
+
})
|
|
292
|
+
|
|
293
|
+
speaker.on("error", (e: any) => {
|
|
294
|
+
reject(e)
|
|
295
|
+
})
|
|
296
|
+
|
|
297
|
+
const bytesPerSecond = rawAudio.sampleRate * 2 * channelCount
|
|
298
|
+
|
|
299
|
+
const byteCountToDuration = (byteCount: number) => {
|
|
300
|
+
return byteCount / bytesPerSecond
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
const audioDuration = byteCountToDuration(audioData.length)
|
|
304
|
+
|
|
305
|
+
let mpg123AudioBufferSize: number
|
|
306
|
+
let mpg123AudioBufferDuration: number
|
|
307
|
+
|
|
308
|
+
if (process.platform == "win32") {
|
|
309
|
+
mpg123AudioBufferSize = 65536
|
|
310
|
+
mpg123AudioBufferDuration = byteCountToDuration(mpg123AudioBufferSize)
|
|
311
|
+
} else {
|
|
312
|
+
mpg123AudioBufferDuration = 0.5
|
|
313
|
+
mpg123AudioBufferSize = bytesPerSecond * mpg123AudioBufferDuration
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
audioData = Buffer.concat([audioData, Buffer.alloc(mpg123AudioBufferSize)])
|
|
317
|
+
|
|
318
|
+
const maxChunkSize = mpg123AudioBufferSize
|
|
319
|
+
|
|
320
|
+
const writeAheadDuration = 0.5
|
|
321
|
+
|
|
322
|
+
const timer = new Timer()
|
|
323
|
+
let readOffset = 0
|
|
324
|
+
let targetTimePosition = 0
|
|
325
|
+
|
|
326
|
+
while (true) {
|
|
327
|
+
const elapsedTime = timer.elapsedTimeSeconds
|
|
328
|
+
|
|
329
|
+
if (onTimePosition) {
|
|
330
|
+
onTimePosition(elapsedTime)
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
if (readOffset < audioData.length) {
|
|
334
|
+
const targetWriteTime = targetTimePosition - writeAheadDuration
|
|
335
|
+
|
|
336
|
+
if (elapsedTime >= targetWriteTime) {
|
|
337
|
+
const chunk = audioData.subarray(readOffset, readOffset + maxChunkSize)
|
|
338
|
+
|
|
339
|
+
speaker.write(chunk)
|
|
340
|
+
|
|
341
|
+
readOffset += chunk.length
|
|
342
|
+
targetTimePosition += byteCountToDuration(chunk.length)
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
if (elapsedTime >= audioDuration) {
|
|
347
|
+
//speaker.close(false)
|
|
348
|
+
resolve()
|
|
349
|
+
return
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
await waitTimeout(20)
|
|
353
|
+
}
|
|
354
|
+
})
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
export const charactersToWriteAhead =
|
|
358
|
+
[",", ".", ",", "、", ":", ";", "。", ":", ";", "?", "!", ")", "]", "}", "\"", "'", "”", "’", "-", "—", "»"]
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import { spawn } from 'child_process'
|
|
2
|
+
|
|
3
|
+
import { RawAudio, concatAudioSegments } from "./AudioUtilities.js"
|
|
4
|
+
import * as AudioBufferConversion from './AudioBufferConversion.js'
|
|
5
|
+
|
|
6
|
+
import { Timer } from "../utilities/Timer.js"
|
|
7
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
8
|
+
import { OpenPromise } from '../utilities/OpenPromise.js'
|
|
9
|
+
import { SampleFormat } from '../codecs/WaveCodec.js'
|
|
10
|
+
import { tryResolvingSoxPath } from './SoxPath.js'
|
|
11
|
+
|
|
12
|
+
const log = logToStderr
|
|
13
|
+
|
|
14
|
+
export async function recordAudioInput(channelCount = 1, sampleRate = 48000, maxTime = 10) {
|
|
15
|
+
const audioSegments: Float32Array[][] = []
|
|
16
|
+
|
|
17
|
+
await captureAudioInput(channelCount, sampleRate, maxTime, (rawAudio) => {
|
|
18
|
+
audioSegments.push(rawAudio)
|
|
19
|
+
})
|
|
20
|
+
|
|
21
|
+
const mergedAudio = concatAudioSegments(audioSegments)
|
|
22
|
+
|
|
23
|
+
const rawAudio: RawAudio = {
|
|
24
|
+
audioChannels: mergedAudio,
|
|
25
|
+
sampleRate
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
return rawAudio
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function captureAudioInput(channelCount = 1, sampleRate = 48000, maxTime = -1, onAudioSamples?: (rawAudio: Float32Array[]) => void) {
|
|
32
|
+
return new Promise<void>(async (resolve, reject) => {
|
|
33
|
+
const timer = new Timer()
|
|
34
|
+
|
|
35
|
+
const recorderSpawnedOpenPromise = new OpenPromise<null>()
|
|
36
|
+
|
|
37
|
+
const soxPath = await tryResolvingSoxPath()
|
|
38
|
+
|
|
39
|
+
if (!soxPath) {
|
|
40
|
+
throw new Error("Could not resolve a SoX executable")
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
let args: string[]
|
|
44
|
+
|
|
45
|
+
if (process.platform == 'win32') {
|
|
46
|
+
args = ['-b', `${16}`, '--endian', 'little',
|
|
47
|
+
'-c', `${channelCount}`, '-r', `${sampleRate}`, '-e', 'signed',
|
|
48
|
+
'-t', 'waveaudio', 'default', '-t', 'raw', '-']
|
|
49
|
+
} else if (process.platform == 'darwin') {
|
|
50
|
+
args = ['-b', `${16}`, '--endian', 'little',
|
|
51
|
+
'-c', `${channelCount}`, '-r', `${sampleRate}`, '-e', 'signed',
|
|
52
|
+
'-t', 'raw', '-']
|
|
53
|
+
} else {
|
|
54
|
+
throw new Error("")
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const recorder = spawn(
|
|
58
|
+
soxPath,
|
|
59
|
+
args,
|
|
60
|
+
{}
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
recorder.once("spawn", () => {
|
|
64
|
+
recorderSpawnedOpenPromise.resolve(null)
|
|
65
|
+
timer.restart()
|
|
66
|
+
})
|
|
67
|
+
|
|
68
|
+
recorder.once('close', () => {
|
|
69
|
+
resolve()
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
recorder.stderr.on('data', (chunk) => {
|
|
73
|
+
//log(chunk.toString('utf8'))
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
recorder.stdout.on('data', (chunk) => {
|
|
77
|
+
//log(chunk)
|
|
78
|
+
|
|
79
|
+
const audioChannels = AudioBufferConversion.decodeToChannels(chunk, channelCount, 16, SampleFormat.PCM)
|
|
80
|
+
|
|
81
|
+
if (onAudioSamples) {
|
|
82
|
+
onAudioSamples(audioChannels)
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
if (maxTime >= 0 && timer.elapsedTimeSeconds >= maxTime) {
|
|
86
|
+
recorder.kill()
|
|
87
|
+
resolve()
|
|
88
|
+
}
|
|
89
|
+
})
|
|
90
|
+
})
|
|
91
|
+
}
|