echogarden 0.11.12 → 0.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +16 -0
- package/dist/api/Alignment.js +2 -2
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +2 -2
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +5 -4
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +2 -2
- package/dist/api/Translation.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +1 -0
- package/dist/audio/AudioUtilities.js +25 -7
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.js +2 -2
- package/dist/cli/CLI.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +2 -2
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/subtitles/Subtitles.d.ts +10 -7
- package/dist/subtitles/Subtitles.js +268 -207
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/docs/Options.md +4 -2
- package/package.json +7 -6
- package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
- package/src/alignment/DTWSequenceAlignment.ts +121 -0
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
- package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
- package/src/alignment/SpeechAlignment.ts +488 -0
- package/src/api/API.ts +12 -0
- package/src/api/APIOptions.ts +15 -0
- package/src/api/Alignment.ts +329 -0
- package/src/api/Common.ts +16 -0
- package/src/api/Denoising.ts +120 -0
- package/src/api/LanguageDetection.ts +286 -0
- package/src/api/Recognition.ts +344 -0
- package/src/api/Synthesis.ts +1735 -0
- package/src/api/Translation.ts +143 -0
- package/src/api/Vad.ts +172 -0
- package/src/audio/AudioBufferConversion.ts +248 -0
- package/src/audio/AudioPlayer.ts +358 -0
- package/src/audio/AudioRecorder.ts +91 -0
- package/src/audio/AudioUtilities.ts +392 -0
- package/src/audio/SoxPath.ts +24 -0
- package/src/cli/CLI.ts +1360 -0
- package/src/cli/CLIConfigFile.ts +91 -0
- package/src/cli/CLILauncher.ts +26 -0
- package/src/cli/CLIOptionsSchema.ts +54 -0
- package/src/cli/CLIParser.ts +41 -0
- package/src/cli/CLIStarter.ts +40 -0
- package/src/codecs/FFMpegTranscoder.ts +214 -0
- package/src/codecs/TIMITCodec.ts +17 -0
- package/src/codecs/WaveCodec.ts +260 -0
- package/src/denoising/RNNoise.ts +95 -0
- package/src/dsp/BiquadFilter.ts +488 -0
- package/src/dsp/FFT.ts +187 -0
- package/src/dsp/MFCC.ts +227 -0
- package/src/dsp/MelSpectogram.ts +145 -0
- package/src/dsp/Rubberband.ts +249 -0
- package/src/dsp/Sonic.ts +59 -0
- package/src/dsp/SpeexResampler.ts +79 -0
- package/src/math/VectorMath.ts +812 -0
- package/src/nlp/ChineseSegmentation.ts +68 -0
- package/src/nlp/CompromiseNLP.ts +113 -0
- package/src/nlp/EspeakPhonemizer.ts +168 -0
- package/src/nlp/IPA.ts +139 -0
- package/src/nlp/JapaneseSegmentation.ts +53 -0
- package/src/nlp/Lexicon.ts +119 -0
- package/src/nlp/PhoneConversion.ts +508 -0
- package/src/nlp/Segmentation.ts +237 -0
- package/src/nlp/TextNormalizer.ts +160 -0
- package/src/recognition/AmazonTranscribeSTT.ts +112 -0
- package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
- package/src/recognition/GoogleCloudSTT.ts +92 -0
- package/src/recognition/SileroSTT.ts +173 -0
- package/src/recognition/VoskSTT.ts +112 -0
- package/src/recognition/WhisperSTT.ts +1518 -0
- package/src/server/Client.ts +297 -0
- package/src/server/Server.ts +178 -0
- package/src/server/ServerStarter.ts +12 -0
- package/src/server/Worker.ts +400 -0
- package/src/server/WorkerStarter.ts +38 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
- package/src/subtitles/Subtitles.ts +478 -0
- package/src/synthesis/AwsPollyTTS.ts +78 -0
- package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
- package/src/synthesis/CoquiServerTTS.ts +29 -0
- package/src/synthesis/ElevenLabsTTS.ts +104 -0
- package/src/synthesis/EspeakTTS.ts +552 -0
- package/src/synthesis/FliteTTS.ts +387 -0
- package/src/synthesis/GoogleCloudTTS.ts +112 -0
- package/src/synthesis/GoogleTranslateTTS.ts +210 -0
- package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
- package/src/synthesis/SamTTS.ts +30 -0
- package/src/synthesis/SapiTTS.ts +222 -0
- package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
- package/src/synthesis/SvoxPicoTTS.ts +318 -0
- package/src/synthesis/VitsTTS.ts +734 -0
- package/src/tests/Test.ts +24 -0
- package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
- package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
- package/src/typings/Fillers.d.ts +41 -0
- package/src/utilities/BinaryArrayConversion.ts +159 -0
- package/src/utilities/Compression.ts +91 -0
- package/src/utilities/FileDownloader.ts +201 -0
- package/src/utilities/FileSystem.ts +265 -0
- package/src/utilities/Hashing.ts +230 -0
- package/src/utilities/Locale.ts +119 -0
- package/src/utilities/Logger.ts +72 -0
- package/src/utilities/NdArrayUtilities.ts +31 -0
- package/src/utilities/ObjectUtilities.ts +169 -0
- package/src/utilities/OpenPromise.ts +13 -0
- package/src/utilities/PackageManager.ts +97 -0
- package/src/utilities/Queue.ts +17 -0
- package/src/utilities/RandomGenerator.ts +237 -0
- package/src/utilities/SignalChannel.ts +22 -0
- package/src/utilities/TarballMaker.ts +68 -0
- package/src/utilities/Timeline.ts +231 -0
- package/src/utilities/Timer.ts +93 -0
- package/src/utilities/Utilities.ts +574 -0
- package/src/utilities/WasmMemoryManager.ts +516 -0
- package/src/utilities/WebReader.ts +55 -0
- package/src/utilities/WikipediaReader.ts +41 -0
- package/src/voice-activity-detection/SileroVAD.ts +86 -0
- package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
import * as AudioBufferConversion from '../audio/AudioBufferConversion.js'
|
|
2
|
+
import { RawAudio } from '../audio/AudioUtilities.js'
|
|
3
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
4
|
+
|
|
5
|
+
const log = logToStderr
|
|
6
|
+
|
|
7
|
+
export function encodeWave(rawAudio: RawAudio, bitDepth: BitDepth = 16, sampleFormat: SampleFormat = SampleFormat.PCM, speakerPositionMask = 0) {
|
|
8
|
+
const audioChannels = rawAudio.audioChannels
|
|
9
|
+
const sampleRate = rawAudio.sampleRate
|
|
10
|
+
|
|
11
|
+
const audioBuffer = AudioBufferConversion.encodeToAudioBuffer(audioChannels, bitDepth, sampleFormat)
|
|
12
|
+
const audioDataLength = audioBuffer.length
|
|
13
|
+
|
|
14
|
+
const shouldUseExtensibleFormat = bitDepth > 16 || audioChannels.length > 2
|
|
15
|
+
|
|
16
|
+
const formatSubChunk = new WaveFormat(audioChannels.length, sampleRate, bitDepth, sampleFormat, speakerPositionMask)
|
|
17
|
+
const formatSubChunkBuffer = formatSubChunk.serialize(shouldUseExtensibleFormat)
|
|
18
|
+
|
|
19
|
+
const dataSubChunkBuffer = Buffer.alloc(4 + 4 + audioDataLength)
|
|
20
|
+
dataSubChunkBuffer.write("data", 0, "ascii")
|
|
21
|
+
dataSubChunkBuffer.writeUint32LE(audioDataLength, 4)
|
|
22
|
+
dataSubChunkBuffer.set(audioBuffer, 8)
|
|
23
|
+
|
|
24
|
+
const riffChunkHeaderBuffer = Buffer.alloc(12)
|
|
25
|
+
riffChunkHeaderBuffer.write("RIFF", 0, "ascii")
|
|
26
|
+
riffChunkHeaderBuffer.writeUint32LE(4 + formatSubChunkBuffer.length + dataSubChunkBuffer.length, 4)
|
|
27
|
+
riffChunkHeaderBuffer.write("WAVE", 8, "ascii")
|
|
28
|
+
|
|
29
|
+
return Buffer.concat([riffChunkHeaderBuffer, formatSubChunkBuffer, dataSubChunkBuffer])
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export function decodeWave(waveData: Buffer, ignoreTruncatedChunks = false) {
|
|
33
|
+
let readOffset = 0
|
|
34
|
+
|
|
35
|
+
const riffId = waveData.subarray(readOffset, readOffset + 4).toString("ascii")
|
|
36
|
+
|
|
37
|
+
if (riffId != "RIFF") {
|
|
38
|
+
throw new Error("Not a valid wave file. No RIFF id found at offset 0.")
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
readOffset += 4
|
|
42
|
+
|
|
43
|
+
const riffChunkSize = waveData.readUInt32LE(readOffset)
|
|
44
|
+
|
|
45
|
+
readOffset += 4
|
|
46
|
+
|
|
47
|
+
const waveId = waveData.subarray(readOffset, readOffset + 4).toString("ascii")
|
|
48
|
+
|
|
49
|
+
if (waveId != "WAVE") {
|
|
50
|
+
throw new Error("Not a valid wave file. No WAVE id found at offset 8.")
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
if (riffChunkSize < waveData.length - 8) {
|
|
54
|
+
throw new Error(`RIFF chunk length ${riffChunkSize} is smaller than the remaining size of the buffer (${waveData.length - 8})`)
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (!ignoreTruncatedChunks && riffChunkSize > waveData.length - 8) {
|
|
58
|
+
throw new Error(`RIFF chunk length (${riffChunkSize}) is greater than the remaining size of the buffer (${waveData.length - 8})`)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
readOffset += 4
|
|
62
|
+
|
|
63
|
+
let formatSubChunkBodyBuffer: Buffer | undefined
|
|
64
|
+
const dataBuffers: Buffer[] = []
|
|
65
|
+
|
|
66
|
+
while (true) {
|
|
67
|
+
const subChunkIdentifier = waveData.subarray(readOffset, readOffset + 4).toString("ascii")
|
|
68
|
+
readOffset += 4
|
|
69
|
+
|
|
70
|
+
const subChunkSize = waveData.readUInt32LE(readOffset)
|
|
71
|
+
readOffset += 4
|
|
72
|
+
|
|
73
|
+
if (!ignoreTruncatedChunks && subChunkSize > waveData.length - readOffset) {
|
|
74
|
+
throw new Error(`Encountered a '${subChunkIdentifier}' subchunk with a size of ${subChunkSize} which is greater than the remaining size of the buffer (${waveData.length - readOffset})`)
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
if (subChunkIdentifier == "fmt ") {
|
|
78
|
+
formatSubChunkBodyBuffer = waveData.subarray(readOffset, readOffset + subChunkSize)
|
|
79
|
+
} else if (subChunkIdentifier == "data") {
|
|
80
|
+
if (!formatSubChunkBodyBuffer) {
|
|
81
|
+
throw new Error("A data subchunk was encountered before a format subchunk")
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// If the data chunk is truncated, but truncations are ignored,
|
|
85
|
+
// it would be read up to the end of the buffer
|
|
86
|
+
dataBuffers.push(waveData.subarray(readOffset, readOffset + subChunkSize))
|
|
87
|
+
}
|
|
88
|
+
// All sub chunks other than 'data' (e.g. 'LIST', 'fact', 'plst', 'junk' etc.) are ignored
|
|
89
|
+
|
|
90
|
+
// This addition operation may overflow if JavaScript integers were 32 bits,
|
|
91
|
+
// but since they are 52 bits, it is okay:
|
|
92
|
+
readOffset += subChunkSize
|
|
93
|
+
|
|
94
|
+
// Break if readOffset is equal to or is greater than the size of the buffer
|
|
95
|
+
if (readOffset >= waveData.length) {
|
|
96
|
+
break
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
if (!formatSubChunkBodyBuffer) {
|
|
101
|
+
throw new Error("No format subchunk was found in the wave file")
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const waveFormat = WaveFormat.deserializeFrom(formatSubChunkBodyBuffer)
|
|
105
|
+
|
|
106
|
+
const sampleFormat = waveFormat.sampleFormat
|
|
107
|
+
const channelCount = waveFormat.channelCount
|
|
108
|
+
const sampleRate = waveFormat.sampleRate
|
|
109
|
+
const bitDepth = waveFormat.bitDepth
|
|
110
|
+
const speakerPositionMask = waveFormat.speakerPositionMask
|
|
111
|
+
|
|
112
|
+
const concatenatedDataBuffers = Buffer.concat(dataBuffers)
|
|
113
|
+
const audioChannels = AudioBufferConversion.decodeToChannels(concatenatedDataBuffers, channelCount, bitDepth, sampleFormat)
|
|
114
|
+
|
|
115
|
+
return {
|
|
116
|
+
rawAudio: { audioChannels, sampleRate },
|
|
117
|
+
|
|
118
|
+
sourceSampleFormat: sampleFormat,
|
|
119
|
+
sourceBitDepth: bitDepth,
|
|
120
|
+
sourceSpeakerPositionMask: speakerPositionMask
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export function repairWave(waveData: Buffer) {
|
|
125
|
+
const { rawAudio, sourceSampleFormat, sourceBitDepth } = decodeWave(waveData)
|
|
126
|
+
|
|
127
|
+
return encodeWave(rawAudio, sourceBitDepth, sourceSampleFormat)
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
class WaveFormat { // 24 bytes total for PCM, 26 for float
|
|
131
|
+
sampleFormat: SampleFormat // 2 bytes LE
|
|
132
|
+
channelCount: number // 2 bytes LE
|
|
133
|
+
sampleRate: number // 4 bytes LE
|
|
134
|
+
get byteRate() { return this.sampleRate * this.bytesPerSample * this.channelCount } // 4 bytes LE
|
|
135
|
+
get blockAlign() { return this.bytesPerSample * this.channelCount } // 2 bytes LE
|
|
136
|
+
bitDepth: BitDepth // 2 bytes LE
|
|
137
|
+
|
|
138
|
+
speakerPositionMask: number // 4 bytes LE
|
|
139
|
+
get guid() { return sampleFormatToGuid[this.sampleFormat] } // 16 bytes BE
|
|
140
|
+
|
|
141
|
+
// helpers:
|
|
142
|
+
get bytesPerSample() { return this.bitDepth / 8 }
|
|
143
|
+
|
|
144
|
+
constructor(channelCount: number, sampleRate: number, bitDepth: BitDepth, sampleFormat: SampleFormat, speakerPositionMask = 0) {
|
|
145
|
+
this.sampleFormat = sampleFormat
|
|
146
|
+
this.channelCount = channelCount
|
|
147
|
+
this.sampleRate = sampleRate
|
|
148
|
+
this.bitDepth = bitDepth
|
|
149
|
+
|
|
150
|
+
this.speakerPositionMask = speakerPositionMask
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
serialize(useExtensibleFormat: boolean) {
|
|
154
|
+
let sampleFormatId = this.sampleFormat
|
|
155
|
+
|
|
156
|
+
if (useExtensibleFormat) {
|
|
157
|
+
sampleFormatId = 65534 as number
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
const serializedSize = sampleFormatToSerializedSize[sampleFormatId]
|
|
161
|
+
|
|
162
|
+
const result = Buffer.alloc(serializedSize)
|
|
163
|
+
|
|
164
|
+
result.write("fmt ", 0, "ascii") // + 4
|
|
165
|
+
result.writeUint32LE(serializedSize - 8, 4) // + 4
|
|
166
|
+
|
|
167
|
+
result.writeUint16LE(sampleFormatId, 8) // + 2
|
|
168
|
+
result.writeUint16LE(this.channelCount, 10) // + 2
|
|
169
|
+
result.writeUint32LE(this.sampleRate, 12) // + 4
|
|
170
|
+
result.writeUint32LE(this.byteRate, 16) // + 4
|
|
171
|
+
result.writeUint16LE(this.blockAlign, 20) // + 2
|
|
172
|
+
result.writeUint16LE(this.bitDepth, 22) // + 2
|
|
173
|
+
|
|
174
|
+
if (useExtensibleFormat) {
|
|
175
|
+
result.writeUint16LE(serializedSize - 26, 24) // + 2 (extension size)
|
|
176
|
+
result.writeUint16LE(this.bitDepth, 26) // + 2 (valid bits per sample)
|
|
177
|
+
result.writeUint32LE(this.speakerPositionMask, 28) // + 2 (speaker position mask)
|
|
178
|
+
|
|
179
|
+
if (this.sampleFormat == SampleFormat.PCM || this.sampleFormat == SampleFormat.Float) {
|
|
180
|
+
result.set(Buffer.from(this.guid, "hex"), 32)
|
|
181
|
+
} else {
|
|
182
|
+
throw new Error(`Extensible format is not supported for sample format ${this.sampleFormat}`)
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
return result
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
static deserializeFrom(formatChunkBody: Buffer) { // chunkBody should not include the first 8 bytes
|
|
190
|
+
let sampleFormat = formatChunkBody.readUint16LE(0) // + 2
|
|
191
|
+
const channelCount = formatChunkBody.readUint16LE(2) // + 2
|
|
192
|
+
const sampleRate = formatChunkBody.readUint32LE(4) // + 4
|
|
193
|
+
const bitDepth = formatChunkBody.readUint16LE(14)
|
|
194
|
+
let speakerPositionMask = 0
|
|
195
|
+
|
|
196
|
+
if (sampleFormat == 65534) {
|
|
197
|
+
if (formatChunkBody.length < 40) {
|
|
198
|
+
throw new Error(`Format subchunk specifies a format id of 65534 (extensible) but its body size is ${formatChunkBody.length} bytes, which is smaller than the minimum expected of 40 bytes`)
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
speakerPositionMask = formatChunkBody.readUint16LE(20)
|
|
202
|
+
|
|
203
|
+
const guid = formatChunkBody.subarray(24, 40).toString("hex")
|
|
204
|
+
|
|
205
|
+
if (guid == sampleFormatToGuid[SampleFormat.PCM]) {
|
|
206
|
+
sampleFormat = SampleFormat.PCM
|
|
207
|
+
} else if (guid == sampleFormatToGuid[SampleFormat.Float]) {
|
|
208
|
+
sampleFormat = SampleFormat.Float
|
|
209
|
+
} else {
|
|
210
|
+
throw new Error(`Unsupported format GUID in extended format subchunk: ${guid}`)
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (sampleFormat == SampleFormat.PCM) {
|
|
215
|
+
if (bitDepth != 8 && bitDepth != 16 && bitDepth != 24 && bitDepth != 32) {
|
|
216
|
+
throw new Error(`PCM audio has a bit depth of ${bitDepth}, which is not supported`)
|
|
217
|
+
}
|
|
218
|
+
} else if (sampleFormat == SampleFormat.Float) {
|
|
219
|
+
if (bitDepth != 32 && bitDepth != 64) {
|
|
220
|
+
throw new Error(`IEEE float audio has a bit depth of ${bitDepth}, which is not supported`)
|
|
221
|
+
}
|
|
222
|
+
} else if (sampleFormat == SampleFormat.Alaw) {
|
|
223
|
+
if (bitDepth != 8) {
|
|
224
|
+
throw new Error(`Alaw audio has a bit depth of ${bitDepth}, which is not supported`)
|
|
225
|
+
}
|
|
226
|
+
} else if (sampleFormat == SampleFormat.Mulaw) {
|
|
227
|
+
if (bitDepth != 8) {
|
|
228
|
+
throw new Error(`Mulaw audio has a bit depth of ${bitDepth}, which is not supported`)
|
|
229
|
+
}
|
|
230
|
+
} else {
|
|
231
|
+
throw new Error(`Wave audio format id ${sampleFormat} is not supported`)
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
return new WaveFormat(channelCount, sampleRate, bitDepth, sampleFormat, speakerPositionMask)
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
export enum SampleFormat {
|
|
239
|
+
PCM = 1,
|
|
240
|
+
Float = 2,
|
|
241
|
+
Alaw = 6,
|
|
242
|
+
Mulaw = 7,
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
export type BitDepth = 8 | 16 | 24 | 32 | 64
|
|
246
|
+
|
|
247
|
+
const sampleFormatToSerializedSize = {
|
|
248
|
+
[SampleFormat.PCM]: 24,
|
|
249
|
+
[SampleFormat.Float]: 26,
|
|
250
|
+
[SampleFormat.Alaw]: 26,
|
|
251
|
+
[SampleFormat.Mulaw]: 26,
|
|
252
|
+
65534: 48
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
const sampleFormatToGuid = {
|
|
256
|
+
[SampleFormat.PCM]: "0100000000001000800000aa00389b71",
|
|
257
|
+
[SampleFormat.Float]: "0300000000001000800000aa00389b71",
|
|
258
|
+
[SampleFormat.Alaw]: "",
|
|
259
|
+
[SampleFormat.Mulaw]: "",
|
|
260
|
+
}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import { float32ToInt16Pcm } from "../audio/AudioBufferConversion.js"
|
|
2
|
+
import { concatFloat32Arrays } from "../utilities/Utilities.js"
|
|
3
|
+
import { WasmMemoryManager } from "../utilities/WasmMemoryManager.js"
|
|
4
|
+
import { Logger } from "../utilities/Logger.js"
|
|
5
|
+
import { RawAudio, cloneRawAudio } from "../audio/AudioUtilities.js"
|
|
6
|
+
|
|
7
|
+
let rnnoiseInstance: any
|
|
8
|
+
|
|
9
|
+
export async function denoiseAudio(rawAudio: RawAudio) {
|
|
10
|
+
const logger = new Logger()
|
|
11
|
+
if (rawAudio.sampleRate != 48000) {
|
|
12
|
+
throw new Error("Sample rate must be 48000")
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
if (rawAudio.audioChannels.length != 1) {
|
|
16
|
+
throw new Error("Channel count must be 1")
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
if (rawAudio.audioChannels[0].length == 0) {
|
|
20
|
+
return { denoisedRawAudio: cloneRawAudio(rawAudio), frameVadProbabilities: [] }
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
logger.start("Get RNNoise WASM instance")
|
|
24
|
+
const m = await getRnnoiseInstance()
|
|
25
|
+
|
|
26
|
+
logger.start("Process with RNNoise")
|
|
27
|
+
const wasmMemory = new WasmMemoryManager(m)
|
|
28
|
+
|
|
29
|
+
const stateSize = m._rnnoise_get_size()
|
|
30
|
+
const frameSize = m._rnnoise_get_frame_size()
|
|
31
|
+
|
|
32
|
+
const denoiseState = m._rnnoise_create(0)
|
|
33
|
+
|
|
34
|
+
const inputRef = wasmMemory.allocFloat32Array(frameSize)
|
|
35
|
+
const outputRef = wasmMemory.allocFloat32Array(frameSize)
|
|
36
|
+
|
|
37
|
+
const floatSamples = rawAudio.audioChannels[0]
|
|
38
|
+
const int16Samples = float32ToInt16Pcm(floatSamples)
|
|
39
|
+
const int16SamplesAsFloats = new Float32Array(int16Samples)
|
|
40
|
+
|
|
41
|
+
const processedFrames: Float32Array[] = []
|
|
42
|
+
const frameVadProbabilities: number[] = []
|
|
43
|
+
|
|
44
|
+
function outputNewFrame(newFrame: Float32Array, vadProbability: number) {
|
|
45
|
+
processedFrames.push(newFrame)
|
|
46
|
+
frameVadProbabilities.push(vadProbability)
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
for (let readOffset = 0; readOffset < int16Samples.length; readOffset += frameSize) {
|
|
50
|
+
let frame = int16SamplesAsFloats.subarray(readOffset, readOffset + frameSize)
|
|
51
|
+
|
|
52
|
+
if (frame.length < frameSize) {
|
|
53
|
+
frame = concatFloat32Arrays([frame, new Float32Array(frameSize - frame.length)])
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
inputRef.view.set(frame)
|
|
57
|
+
|
|
58
|
+
const vadProbability = m._rnnoise_process_frame(denoiseState, outputRef.address, inputRef.address)
|
|
59
|
+
|
|
60
|
+
// Latency compensation: don't write an output frame for the first read frame
|
|
61
|
+
if (readOffset > 0) {
|
|
62
|
+
outputNewFrame(outputRef.view.slice(), vadProbability)
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
// Latency compensation: process an empty input frame for the last output frame
|
|
67
|
+
inputRef.view.set(new Float32Array(frameSize))
|
|
68
|
+
|
|
69
|
+
const lastFrameVadProbability = m._rnnoise_process_frame(denoiseState, outputRef.address, inputRef.address)
|
|
70
|
+
outputNewFrame(outputRef.view.slice(), lastFrameVadProbability)
|
|
71
|
+
|
|
72
|
+
m._rnnoise_destroy(denoiseState)
|
|
73
|
+
wasmMemory.freeAll()
|
|
74
|
+
|
|
75
|
+
const int16DenoisedSamplesAsFloats = concatFloat32Arrays(processedFrames)
|
|
76
|
+
|
|
77
|
+
let denoisedSamples = int16DenoisedSamplesAsFloats.map(sample => sample / 32768)
|
|
78
|
+
denoisedSamples = denoisedSamples.subarray(0, floatSamples.length)
|
|
79
|
+
|
|
80
|
+
const denoisedRawAudio: RawAudio = { audioChannels: [denoisedSamples], sampleRate: 48000 }
|
|
81
|
+
|
|
82
|
+
logger.end()
|
|
83
|
+
|
|
84
|
+
return { denoisedRawAudio, frameVadProbabilities }
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export async function getRnnoiseInstance() {
|
|
88
|
+
if (!rnnoiseInstance) {
|
|
89
|
+
const { default: initializer } = await import('@echogarden/rnnoise-wasm')
|
|
90
|
+
|
|
91
|
+
rnnoiseInstance = await initializer()
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return rnnoiseInstance
|
|
95
|
+
}
|