echogarden 0.11.12 → 0.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +16 -0
- package/dist/api/Alignment.js +2 -2
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +2 -2
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +5 -4
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +2 -2
- package/dist/api/Translation.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +1 -0
- package/dist/audio/AudioUtilities.js +25 -7
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.js +2 -2
- package/dist/cli/CLI.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +2 -2
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/subtitles/Subtitles.d.ts +10 -7
- package/dist/subtitles/Subtitles.js +268 -207
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/docs/Options.md +4 -2
- package/package.json +7 -6
- package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
- package/src/alignment/DTWSequenceAlignment.ts +121 -0
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
- package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
- package/src/alignment/SpeechAlignment.ts +488 -0
- package/src/api/API.ts +12 -0
- package/src/api/APIOptions.ts +15 -0
- package/src/api/Alignment.ts +329 -0
- package/src/api/Common.ts +16 -0
- package/src/api/Denoising.ts +120 -0
- package/src/api/LanguageDetection.ts +286 -0
- package/src/api/Recognition.ts +344 -0
- package/src/api/Synthesis.ts +1735 -0
- package/src/api/Translation.ts +143 -0
- package/src/api/Vad.ts +172 -0
- package/src/audio/AudioBufferConversion.ts +248 -0
- package/src/audio/AudioPlayer.ts +358 -0
- package/src/audio/AudioRecorder.ts +91 -0
- package/src/audio/AudioUtilities.ts +392 -0
- package/src/audio/SoxPath.ts +24 -0
- package/src/cli/CLI.ts +1360 -0
- package/src/cli/CLIConfigFile.ts +91 -0
- package/src/cli/CLILauncher.ts +26 -0
- package/src/cli/CLIOptionsSchema.ts +54 -0
- package/src/cli/CLIParser.ts +41 -0
- package/src/cli/CLIStarter.ts +40 -0
- package/src/codecs/FFMpegTranscoder.ts +214 -0
- package/src/codecs/TIMITCodec.ts +17 -0
- package/src/codecs/WaveCodec.ts +260 -0
- package/src/denoising/RNNoise.ts +95 -0
- package/src/dsp/BiquadFilter.ts +488 -0
- package/src/dsp/FFT.ts +187 -0
- package/src/dsp/MFCC.ts +227 -0
- package/src/dsp/MelSpectogram.ts +145 -0
- package/src/dsp/Rubberband.ts +249 -0
- package/src/dsp/Sonic.ts +59 -0
- package/src/dsp/SpeexResampler.ts +79 -0
- package/src/math/VectorMath.ts +812 -0
- package/src/nlp/ChineseSegmentation.ts +68 -0
- package/src/nlp/CompromiseNLP.ts +113 -0
- package/src/nlp/EspeakPhonemizer.ts +168 -0
- package/src/nlp/IPA.ts +139 -0
- package/src/nlp/JapaneseSegmentation.ts +53 -0
- package/src/nlp/Lexicon.ts +119 -0
- package/src/nlp/PhoneConversion.ts +508 -0
- package/src/nlp/Segmentation.ts +237 -0
- package/src/nlp/TextNormalizer.ts +160 -0
- package/src/recognition/AmazonTranscribeSTT.ts +112 -0
- package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
- package/src/recognition/GoogleCloudSTT.ts +92 -0
- package/src/recognition/SileroSTT.ts +173 -0
- package/src/recognition/VoskSTT.ts +112 -0
- package/src/recognition/WhisperSTT.ts +1518 -0
- package/src/server/Client.ts +297 -0
- package/src/server/Server.ts +178 -0
- package/src/server/ServerStarter.ts +12 -0
- package/src/server/Worker.ts +400 -0
- package/src/server/WorkerStarter.ts +38 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
- package/src/subtitles/Subtitles.ts +478 -0
- package/src/synthesis/AwsPollyTTS.ts +78 -0
- package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
- package/src/synthesis/CoquiServerTTS.ts +29 -0
- package/src/synthesis/ElevenLabsTTS.ts +104 -0
- package/src/synthesis/EspeakTTS.ts +552 -0
- package/src/synthesis/FliteTTS.ts +387 -0
- package/src/synthesis/GoogleCloudTTS.ts +112 -0
- package/src/synthesis/GoogleTranslateTTS.ts +210 -0
- package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
- package/src/synthesis/SamTTS.ts +30 -0
- package/src/synthesis/SapiTTS.ts +222 -0
- package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
- package/src/synthesis/SvoxPicoTTS.ts +318 -0
- package/src/synthesis/VitsTTS.ts +734 -0
- package/src/tests/Test.ts +24 -0
- package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
- package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
- package/src/typings/Fillers.d.ts +41 -0
- package/src/utilities/BinaryArrayConversion.ts +159 -0
- package/src/utilities/Compression.ts +91 -0
- package/src/utilities/FileDownloader.ts +201 -0
- package/src/utilities/FileSystem.ts +265 -0
- package/src/utilities/Hashing.ts +230 -0
- package/src/utilities/Locale.ts +119 -0
- package/src/utilities/Logger.ts +72 -0
- package/src/utilities/NdArrayUtilities.ts +31 -0
- package/src/utilities/ObjectUtilities.ts +169 -0
- package/src/utilities/OpenPromise.ts +13 -0
- package/src/utilities/PackageManager.ts +97 -0
- package/src/utilities/Queue.ts +17 -0
- package/src/utilities/RandomGenerator.ts +237 -0
- package/src/utilities/SignalChannel.ts +22 -0
- package/src/utilities/TarballMaker.ts +68 -0
- package/src/utilities/Timeline.ts +231 -0
- package/src/utilities/Timer.ts +93 -0
- package/src/utilities/Utilities.ts +574 -0
- package/src/utilities/WasmMemoryManager.ts +516 -0
- package/src/utilities/WebReader.ts +55 -0
- package/src/utilities/WikipediaReader.ts +41 -0
- package/src/voice-activity-detection/SileroVAD.ts +86 -0
- package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
|
@@ -0,0 +1,392 @@
|
|
|
1
|
+
import * as FFMpegTranscoder from "../codecs/FFMpegTranscoder.js"
|
|
2
|
+
import { SampleFormat, encodeWave, decodeWave, BitDepth } from "../codecs/WaveCodec.js"
|
|
3
|
+
import { resampleAudioSpeex } from "../dsp/SpeexResampler.js"
|
|
4
|
+
import { concatFloat32Arrays } from '../utilities/Utilities.js'
|
|
5
|
+
|
|
6
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
7
|
+
// Wave encoding and decoding
|
|
8
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
9
|
+
export function encodeWaveBuffer(rawAudio: RawAudio, bitDepth: BitDepth = 16, sampleFormat: SampleFormat = SampleFormat.PCM, speakerPositionMask = 0) {
|
|
10
|
+
return encodeWave(rawAudio, bitDepth, sampleFormat, speakerPositionMask)
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function decodeWaveBuffer(waveFileBuffer: Buffer, ignoreTruncatedChunks = false) {
|
|
14
|
+
return decodeWave(waveFileBuffer, ignoreTruncatedChunks)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
18
|
+
// Audio trimming
|
|
19
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
20
|
+
const defaultSilenceThresholdDecibels = -40
|
|
21
|
+
|
|
22
|
+
export function trimAudioStart(audioSamples: Float32Array, targetStartSilentSampleCount = 0, amplitudeThresholdDecibels = defaultSilenceThresholdDecibels) {
|
|
23
|
+
const silentSampleCount = getStartingSilentSampleCount(audioSamples, amplitudeThresholdDecibels)
|
|
24
|
+
|
|
25
|
+
const trimmedAudio = audioSamples.subarray(silentSampleCount, audioSamples.length)
|
|
26
|
+
const restoredSilence = new Float32Array(targetStartSilentSampleCount)
|
|
27
|
+
|
|
28
|
+
const trimmedAudioSamples = concatFloat32Arrays([restoredSilence, trimmedAudio])
|
|
29
|
+
|
|
30
|
+
return trimmedAudioSamples
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function trimAudioEnd(audioSamples: Float32Array, targetEndSilentSampleCount = 0, amplitudeThresholdDecibels = defaultSilenceThresholdDecibels) {
|
|
34
|
+
const silentSampleCount = getEndingSilentSampleCount(audioSamples, amplitudeThresholdDecibels)
|
|
35
|
+
|
|
36
|
+
const trimmedAudio = audioSamples.subarray(0, audioSamples.length - silentSampleCount)
|
|
37
|
+
const restoredSilence = new Float32Array(targetEndSilentSampleCount)
|
|
38
|
+
|
|
39
|
+
const trimmedAudioSamples = concatFloat32Arrays([trimmedAudio, restoredSilence])
|
|
40
|
+
|
|
41
|
+
return trimmedAudioSamples
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function getStartingSilentSampleCount(audioSamples: Float32Array, amplitudeThresholdDecibels = defaultSilenceThresholdDecibels) {
|
|
45
|
+
const minSampleValue = decibelsToGain(amplitudeThresholdDecibels)
|
|
46
|
+
|
|
47
|
+
let silentSampleCount = 0
|
|
48
|
+
|
|
49
|
+
for (let i = 0; i < audioSamples.length - 1; i++) {
|
|
50
|
+
if (Math.abs(audioSamples[i]) > minSampleValue) {
|
|
51
|
+
break
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
silentSampleCount += 1
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
return silentSampleCount
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function getEndingSilentSampleCount(audioSamples: Float32Array, amplitudeThresholdDecibels = defaultSilenceThresholdDecibels) {
|
|
61
|
+
const minSampleValue = decibelsToGain(amplitudeThresholdDecibels)
|
|
62
|
+
|
|
63
|
+
let silentSampleCount = 0
|
|
64
|
+
|
|
65
|
+
for (let i = audioSamples.length - 1; i >= 0; i--) {
|
|
66
|
+
if (Math.abs(audioSamples[i]) > minSampleValue) {
|
|
67
|
+
break
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
silentSampleCount += 1
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
return silentSampleCount
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
77
|
+
// Gain, normalization, mixing, and channel downmixing
|
|
78
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
79
|
+
export function downmixToMonoAndNormalize(rawAudio: RawAudio, targetPeakDb = -3) {
|
|
80
|
+
return normalizeAudioLevel(downmixToMono(rawAudio), targetPeakDb)
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export function normalizeAudioLevel(rawAudio: RawAudio, targetPeakDb = -3, maxIncreaseDb = 30): RawAudio {
|
|
84
|
+
//rawAudio = correctDCBias(rawAudio)
|
|
85
|
+
|
|
86
|
+
const targetPeakSampleValue = decibelsToGain(targetPeakDb)
|
|
87
|
+
const maxMultiplier = decibelsToGain(maxIncreaseDb)
|
|
88
|
+
|
|
89
|
+
const maxAbsoluteSampleValue = getAudioPeakGain(rawAudio.audioChannels)
|
|
90
|
+
|
|
91
|
+
const multiplier = Math.min(targetPeakSampleValue / maxAbsoluteSampleValue, maxMultiplier)
|
|
92
|
+
|
|
93
|
+
return applyGain(rawAudio, multiplier)
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export function correctDCBias(rawAudio: RawAudio): RawAudio {
|
|
97
|
+
const outputAudioChannels: Float32Array[] = []
|
|
98
|
+
|
|
99
|
+
for (const channelSamples of rawAudio.audioChannels) {
|
|
100
|
+
const sampleCount = channelSamples.length
|
|
101
|
+
|
|
102
|
+
let sampleSum = 0
|
|
103
|
+
|
|
104
|
+
for (let i = 0; i < sampleCount; i++) {
|
|
105
|
+
sampleSum += channelSamples[i]
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const sampleAverage = sampleSum / sampleCount
|
|
109
|
+
|
|
110
|
+
const outputChannelSamples = new Float32Array(sampleCount)
|
|
111
|
+
|
|
112
|
+
for (let i = 0; i < sampleCount; i++) {
|
|
113
|
+
outputChannelSamples[i] = channelSamples[i] - sampleAverage
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
outputAudioChannels.push(outputChannelSamples)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
return { audioChannels: outputAudioChannels, sampleRate: rawAudio.sampleRate } as RawAudio
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
export function applyGainDecibels(rawAudio: RawAudio, decibelGain: number): RawAudio {
|
|
123
|
+
return applyGain(rawAudio, decibelsToGain(decibelGain))
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
export function applyGain(rawAudio: RawAudio, gain: number): RawAudio {
|
|
127
|
+
const outputAudioChannels: Float32Array[] = []
|
|
128
|
+
|
|
129
|
+
const multiplier = gain
|
|
130
|
+
|
|
131
|
+
for (const channelSamples of rawAudio.audioChannels) {
|
|
132
|
+
const sampleCount = channelSamples.length
|
|
133
|
+
|
|
134
|
+
const outputChannelSamples = new Float32Array(sampleCount)
|
|
135
|
+
|
|
136
|
+
for (let i = 0; i < sampleCount; i++) {
|
|
137
|
+
outputChannelSamples[i] = channelSamples[i] * multiplier
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
outputAudioChannels.push(outputChannelSamples)
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
return { audioChannels: outputAudioChannels, sampleRate: rawAudio.sampleRate } as RawAudio
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
export function downmixToMono(rawAudio: RawAudio): RawAudio {
|
|
147
|
+
const sampleCount = rawAudio.audioChannels[0].length
|
|
148
|
+
|
|
149
|
+
const downmixedAudio = new Float32Array(sampleCount)
|
|
150
|
+
|
|
151
|
+
for (const channelSamples of rawAudio.audioChannels) {
|
|
152
|
+
for (let i = 0; i < sampleCount; i++) {
|
|
153
|
+
downmixedAudio[i] += channelSamples[i]
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
return { audioChannels: [downmixedAudio], sampleRate: rawAudio.sampleRate } as RawAudio
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export function getAudioPeakDecibels(audioChannels: Float32Array[]) {
|
|
161
|
+
return gainToDecibels(getAudioPeakGain(audioChannels))
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
export function getAudioPeakGain(audioChannels: Float32Array[]) {
|
|
165
|
+
let maxAbsoluteSampleValue = 0.00001
|
|
166
|
+
|
|
167
|
+
for (const channelSamples of audioChannels) {
|
|
168
|
+
for (const sample of channelSamples) {
|
|
169
|
+
maxAbsoluteSampleValue = Math.max(maxAbsoluteSampleValue, Math.abs(sample))
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return maxAbsoluteSampleValue
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export function mixAudio(rawAudio1: RawAudio, rawAudio2: RawAudio) {
|
|
177
|
+
if (rawAudio1.audioChannels.length != rawAudio2.audioChannels.length) {
|
|
178
|
+
throw new Error("Can't mix audio of unequal channel counts")
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
if (rawAudio1.sampleRate != rawAudio2.sampleRate) {
|
|
182
|
+
throw new Error("Can't mix audio of different sample rates")
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const mixedAudioChannels: Float32Array[] = []
|
|
186
|
+
|
|
187
|
+
for (let c = 0; c < rawAudio1.audioChannels.length; c++) {
|
|
188
|
+
const inputChannel1 = rawAudio1.audioChannels[c]
|
|
189
|
+
const inputChannel2 = rawAudio2.audioChannels[c]
|
|
190
|
+
|
|
191
|
+
const mixedChannelLength = Math.min(inputChannel1.length, inputChannel2.length)
|
|
192
|
+
|
|
193
|
+
const mixedChannel = new Float32Array(mixedChannelLength)
|
|
194
|
+
|
|
195
|
+
for (let i = 0; i < mixedChannelLength; i++) {
|
|
196
|
+
mixedChannel[i] = inputChannel1[i] + inputChannel2[i]
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
mixedAudioChannels.push(mixedChannel)
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const mixedAudio: RawAudio = { audioChannels: mixedAudioChannels, sampleRate: rawAudio1.sampleRate }
|
|
203
|
+
|
|
204
|
+
return mixedAudio
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
208
|
+
// Cutting, concatenation, and other operations
|
|
209
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
210
|
+
export function sliceRawAudioByTime(rawAudio: RawAudio, startTime: number, endTime: number): RawAudio {
|
|
211
|
+
const startSampleIndex = Math.floor(startTime * rawAudio.sampleRate)
|
|
212
|
+
const endSampleIndex = Math.floor(endTime * rawAudio.sampleRate)
|
|
213
|
+
|
|
214
|
+
return sliceRawAudio(rawAudio, startSampleIndex, endSampleIndex)
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
export function sliceRawAudio(rawAudio: RawAudio, startSampleIndex: number, endSampleIndex: number): RawAudio {
|
|
218
|
+
return { audioChannels: sliceAudioChannels(rawAudio.audioChannels, startSampleIndex, endSampleIndex), sampleRate: rawAudio.sampleRate } as RawAudio
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
export function sliceAudioChannels(audioChannels: Float32Array[], startSampleIndex: number, endSampleIndex: number) {
|
|
222
|
+
const channelCount = audioChannels.length
|
|
223
|
+
|
|
224
|
+
const outAudioChannels: Float32Array[] = []
|
|
225
|
+
|
|
226
|
+
for (let i = 0; i < channelCount; i++) {
|
|
227
|
+
outAudioChannels.push(audioChannels[i].slice(startSampleIndex, endSampleIndex))
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
return outAudioChannels
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
export function concatAudioSegments(audioSegments: Float32Array[][]) {
|
|
234
|
+
if (audioSegments.length == 0) {
|
|
235
|
+
return []
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
const channelCount = audioSegments[0].length
|
|
239
|
+
|
|
240
|
+
const outAudioChannels: Float32Array[] = []
|
|
241
|
+
|
|
242
|
+
for (let i = 0; i < channelCount; i++) {
|
|
243
|
+
const audioSegmentsForChannel = audioSegments.map(segment => segment[i])
|
|
244
|
+
|
|
245
|
+
outAudioChannels.push(concatFloat32Arrays(audioSegmentsForChannel))
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
return outAudioChannels
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
export function fadeAudioInOut(rawAudio: RawAudio, fadeTime: number): RawAudio {
|
|
252
|
+
const fadeSampleCount = Math.floor(rawAudio.sampleRate * fadeTime)
|
|
253
|
+
const gainReductionPerFrameDecibels = -60 / fadeSampleCount
|
|
254
|
+
|
|
255
|
+
const gainReductionPerFrameMultiplier = decibelsToGain(gainReductionPerFrameDecibels)
|
|
256
|
+
|
|
257
|
+
const outAudioChannels = rawAudio.audioChannels.map(channel => channel.slice())
|
|
258
|
+
|
|
259
|
+
for (const channel of outAudioChannels) {
|
|
260
|
+
if (channel.length < fadeSampleCount * 2) {
|
|
261
|
+
continue
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
let multiplier = 1.0
|
|
265
|
+
|
|
266
|
+
for (let i = fadeSampleCount - 1; i >= 0; i--) {
|
|
267
|
+
channel[i] *= multiplier
|
|
268
|
+
multiplier *= gainReductionPerFrameMultiplier
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
multiplier = 1.0
|
|
272
|
+
|
|
273
|
+
for (let i = channel.length - fadeSampleCount; i < channel.length; i++) {
|
|
274
|
+
channel[i] *= multiplier
|
|
275
|
+
multiplier *= gainReductionPerFrameMultiplier
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
return { audioChannels: outAudioChannels, sampleRate: rawAudio.sampleRate } as RawAudio
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
export function cloneRawAudio(rawAudio: RawAudio): RawAudio {
|
|
283
|
+
return {
|
|
284
|
+
audioChannels: rawAudio.audioChannels.map(channel => channel.slice()),
|
|
285
|
+
sampleRate: rawAudio.sampleRate
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
export function getSilentAudio(sampleCount: number, channelCount: number) {
|
|
290
|
+
const audioChannels: Float32Array[] = []
|
|
291
|
+
|
|
292
|
+
for (let i = 0; i < channelCount; i++) {
|
|
293
|
+
audioChannels.push(new Float32Array(sampleCount))
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
return audioChannels
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
export function getEmptyRawAudio(channelCount: number, sampleRate: number) {
|
|
300
|
+
const audioChannels = []
|
|
301
|
+
|
|
302
|
+
for (let c = 0; c < channelCount; c++) {
|
|
303
|
+
audioChannels.push(new Float32Array(0))
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
const result: RawAudio = { audioChannels, sampleRate }
|
|
307
|
+
|
|
308
|
+
return result
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
export function getRawAudioDuration(rawAudio: RawAudio) {
|
|
312
|
+
if (rawAudio.audioChannels.length == 0 || rawAudio.sampleRate == 0) {
|
|
313
|
+
return 0
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
return rawAudio.audioChannels[0].length / rawAudio.sampleRate
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
export async function ensureRawAudio(input: AudioSourceParam, outSampleRate?: number, outChannelCount?: number) {
|
|
320
|
+
let inputAsRawAudio: RawAudio = input as RawAudio
|
|
321
|
+
|
|
322
|
+
if (inputAsRawAudio.audioChannels?.length > 0 && inputAsRawAudio.sampleRate) {
|
|
323
|
+
const inputAudioChannelCount = inputAsRawAudio.audioChannels.length
|
|
324
|
+
|
|
325
|
+
if (outChannelCount == 1 && inputAudioChannelCount > 1) {
|
|
326
|
+
inputAsRawAudio = downmixToMono(inputAsRawAudio)
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
if (outChannelCount != null && outChannelCount >= 2 && outChannelCount != inputAudioChannelCount) {
|
|
330
|
+
throw new Error(`Can't convert ${inputAudioChannelCount} channels to ${outChannelCount} channels. Channel conversion of raw audio currently only supports downmixing to mono.`)
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
if (outSampleRate && inputAsRawAudio.sampleRate != outSampleRate) {
|
|
334
|
+
inputAsRawAudio = await resampleAudioSpeex(inputAsRawAudio, outSampleRate)
|
|
335
|
+
}
|
|
336
|
+
} else if (typeof input == "string" || input instanceof Uint8Array) {
|
|
337
|
+
if (input instanceof Uint8Array && !Buffer.isBuffer(input)) {
|
|
338
|
+
input = Buffer.from(input)
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
const inputAsStringOrBuffer = input as string | Buffer
|
|
342
|
+
|
|
343
|
+
inputAsRawAudio = await FFMpegTranscoder.decodeToChannels(inputAsStringOrBuffer, outSampleRate, outChannelCount)
|
|
344
|
+
} else {
|
|
345
|
+
throw new Error("Received an invalid input audio data type.")
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
return inputAsRawAudio
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
352
|
+
// Unit conversion
|
|
353
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
354
|
+
|
|
355
|
+
export function gainToDecibels(gain: number) {
|
|
356
|
+
return gain <= 0.00001 ? -100 : (20.0 * Math.log10(gain))
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
export function decibelsToGain(decibels: number) {
|
|
360
|
+
return decibels <= -100.0 ? 0 : Math.pow(10, 0.05 * decibels)
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
export function powerToDecibels(power: number) {
|
|
364
|
+
return power <= 0.0000000001 ? -100 : (10.0 * Math.log10(power))
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
export function decibelsToPower(decibels: number) {
|
|
368
|
+
return decibels <= -100.0 ? 0 : Math.pow(10, 0.1 * decibels)
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
372
|
+
// Types
|
|
373
|
+
////////////////////////////////////////////////////////////////////////////////////////////////
|
|
374
|
+
|
|
375
|
+
export type RawAudio = {
|
|
376
|
+
audioChannels: Float32Array[]
|
|
377
|
+
sampleRate: number
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
export type AudioEncoding = {
|
|
381
|
+
codec?: string
|
|
382
|
+
format: string
|
|
383
|
+
|
|
384
|
+
channelCount: number
|
|
385
|
+
sampleRate: number
|
|
386
|
+
bitdepth: number
|
|
387
|
+
sampleFormat: SampleFormat
|
|
388
|
+
|
|
389
|
+
bitrate?: number
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
export type AudioSourceParam = string | Buffer | Uint8Array | RawAudio
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
import path from "path"
|
|
2
|
+
import { loadPackage } from "../utilities/PackageManager.js"
|
|
3
|
+
import { commandExists } from "../utilities/Utilities.js"
|
|
4
|
+
|
|
5
|
+
export async function tryResolvingSoxPath() {
|
|
6
|
+
let soxPath: string | undefined = undefined
|
|
7
|
+
|
|
8
|
+
if (process.platform == "win32") {
|
|
9
|
+
const soxPackagePath = await loadPackage("sox-14.4.1a-win32")
|
|
10
|
+
soxPath = path.join(soxPackagePath, "sox.exe")
|
|
11
|
+
}
|
|
12
|
+
/*else if (process.platform == "darwin" && process.arch == "x64") {
|
|
13
|
+
const soxPackagePath = await loadPackage("sox-14.4.1-macosx")
|
|
14
|
+
soxPath = path.join(soxPackagePath, "sox")
|
|
15
|
+
} */
|
|
16
|
+
else if (process.platform == "linux" && process.arch == "x64") {
|
|
17
|
+
const soxPackagePath = await loadPackage("sox-14.4.2-linux-minimal")
|
|
18
|
+
soxPath = path.join(soxPackagePath, "sox")
|
|
19
|
+
} else if (await commandExists("sox")) {
|
|
20
|
+
soxPath = "sox"
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
return soxPath
|
|
24
|
+
}
|