echogarden 1.4.4 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +310 -25
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +4 -2
- package/dist/alignment/SemanticTextAlignment.js +336 -0
- package/dist/alignment/SemanticTextAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +4 -3
- package/dist/alignment/SpeechAlignment.js +130 -39
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +7 -3
- package/dist/api/API.js +7 -2
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +4 -1
- package/dist/api/Alignment.d.ts +1 -1
- package/dist/api/Alignment.js +13 -5
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetectionCommon.d.ts +6 -0
- package/dist/api/LanguageDetectionCommon.js +2 -0
- package/dist/api/LanguageDetectionCommon.js.map +1 -0
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
- package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
- package/dist/api/SpeechLanguageDetection.js.map +1 -0
- package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
- package/dist/api/SpeechTranslation.js.map +1 -0
- package/dist/api/Synthesis.d.ts +0 -1
- package/dist/api/Synthesis.js +4 -4
- package/dist/api/TextLanguageDetection.d.ts +21 -0
- package/dist/api/TextLanguageDetection.js +67 -0
- package/dist/api/TextLanguageDetection.js.map +1 -0
- package/dist/api/TextTranslation.d.ts +25 -0
- package/dist/api/TextTranslation.js +101 -0
- package/dist/api/TextTranslation.js.map +1 -0
- package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
- package/dist/api/TimelineTranslationAlignment.js +92 -0
- package/dist/api/TimelineTranslationAlignment.js.map +1 -0
- package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
- package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
- package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
- package/dist/api/TranslationAlignment.d.ts +4 -3
- package/dist/api/TranslationAlignment.js +9 -8
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.js +16 -1
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioBufferConversion.d.ts +0 -1
- package/dist/audio/AudioPlayer.d.ts +0 -1
- package/dist/audio/AudioPlayer.js +62 -41
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +0 -1
- package/dist/cli/CLI.d.ts +28 -7
- package/dist/cli/CLI.js +265 -37
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
- package/dist/codecs/FFMpegTranscoder.js +7 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/codecs/TIMITCodec.d.ts +0 -1
- package/dist/codecs/WaveCodec.d.ts +0 -1
- package/dist/dsp/FFT.d.ts +1 -1
- package/dist/dsp/FFT.js +6 -0
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.js +1 -1
- package/dist/dsp/KWeightingFilter.js.map +1 -1
- package/dist/dsp/MelSpectogram.d.ts +3 -2
- package/dist/dsp/MelSpectogram.js +14 -8
- package/dist/dsp/MelSpectogram.js.map +1 -1
- package/dist/math/VectorMath.d.ts +9 -9
- package/dist/math/VectorMath.js +10 -10
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/nlp/ChineseSegmentation.js +4 -4
- package/dist/nlp/ChineseSegmentation.js.map +1 -1
- package/dist/nlp/Segmentation.d.ts +2 -2
- package/dist/nlp/Segmentation.js +20 -13
- package/dist/nlp/Segmentation.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
- package/dist/recognition/OpenAICloudSTT.js +30 -19
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +0 -1
- package/dist/recognition/WhisperCppSTT.d.ts +3 -3
- package/dist/recognition/WhisperCppSTT.js +21 -9
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +9 -6
- package/dist/recognition/WhisperSTT.js +227 -46
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +3 -4
- package/dist/server/Client.js.map +1 -1
- package/dist/server/Worker.d.ts +3 -3
- package/dist/server/Worker.js +3 -2
- package/dist/server/Worker.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
- package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +12 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -2
- package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/subtitles/Subtitles.js +2 -2
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
- package/dist/synthesis/GoogleTranslateTTS.js +6 -21
- package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.d.ts +0 -1
- package/dist/synthesis/VitsTTS.js +30 -0
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js +0 -31
- package/dist/tests/Test.js.map +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
- package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
- package/dist/text-translation/DeepLTextTranslation.js +67 -0
- package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
- package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
- package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
- package/dist/text-translation/NLLBTextTranslation.js +249 -19
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
- package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
- package/dist/utilities/BrowserRequestHeaders.js +52 -0
- package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
- package/dist/utilities/BufferFileReadStream.d.ts +20 -0
- package/dist/utilities/BufferFileReadStream.js +81 -0
- package/dist/utilities/BufferFileReadStream.js.map +1 -0
- package/dist/utilities/DynamicUint8Array.d.ts +9 -0
- package/dist/utilities/DynamicUint8Array.js +31 -0
- package/dist/utilities/DynamicUint8Array.js.map +1 -0
- package/dist/utilities/FileSystem.d.ts +0 -2
- package/dist/utilities/Hashing.d.ts +3 -10
- package/dist/utilities/Hashing.js +10 -127
- package/dist/utilities/Hashing.js.map +1 -1
- package/dist/utilities/LEB128.d.ts +15 -5
- package/dist/utilities/LEB128.js +199 -119
- package/dist/utilities/LEB128.js.map +1 -1
- package/dist/utilities/LPVarInt.d.ts +11 -0
- package/dist/utilities/LPVarInt.js +187 -0
- package/dist/utilities/LPVarInt.js.map +1 -0
- package/dist/utilities/Locale.d.ts +1 -1
- package/dist/utilities/Locale.js +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +1 -2
- package/dist/utilities/PVarInt.d.ts +4 -0
- package/dist/utilities/PVarInt.js +166 -0
- package/dist/utilities/PVarInt.js.map +1 -0
- package/dist/utilities/PackageManager.js +48 -25
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/RandomGenerator.d.ts +3 -17
- package/dist/utilities/RandomGenerator.js +12 -81
- package/dist/utilities/RandomGenerator.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +2 -0
- package/dist/utilities/Timeline.js +129 -20
- package/dist/utilities/Timeline.js.map +1 -1
- package/dist/utilities/Utilities.d.ts +1 -3
- package/dist/utilities/Utilities.js +30 -3
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/utilities/VarInt.d.ts +4 -0
- package/dist/utilities/VarInt.js +166 -0
- package/dist/utilities/VarInt.js.map +1 -0
- package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
- package/dist/utilities/VirtualFileReadStream.js +79 -0
- package/dist/utilities/VirtualFileReadStream.js.map +1 -0
- package/dist/utilities/WebReader.js +7 -23
- package/dist/utilities/WebReader.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
- package/docs/API.md +105 -3
- package/docs/CLI.md +51 -1
- package/docs/Engines.md +32 -3
- package/docs/Options.md +53 -12
- package/docs/Tasklist.md +1 -13
- package/package.json +20 -24
- package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
- package/src/alignment/SemanticTextAlignment.ts +467 -0
- package/src/alignment/SpeechAlignment.ts +214 -56
- package/src/api/API.ts +18 -2
- package/src/api/APIOptions.ts +14 -1
- package/src/api/Alignment.ts +31 -9
- package/src/api/LanguageDetectionCommon.ts +7 -0
- package/src/api/Recognition.ts +2 -0
- package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
- package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
- package/src/api/Synthesis.ts +4 -4
- package/src/api/TextLanguageDetection.ts +116 -0
- package/src/api/TextTranslation.ts +177 -0
- package/src/api/TimelineTranslationAlignment.ts +162 -0
- package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
- package/src/api/TranslationAlignment.ts +12 -10
- package/src/api/VoiceActivityDetection.ts +24 -3
- package/src/audio/AudioPlayer.ts +2 -0
- package/src/cli/CLI.ts +376 -40
- package/src/codecs/FFMpegTranscoder.ts +6 -0
- package/src/dsp/FFT.ts +8 -2
- package/src/dsp/KWeightingFilter.ts +1 -1
- package/src/dsp/MelSpectogram.ts +17 -8
- package/src/math/VectorMath.ts +15 -15
- package/src/nlp/ChineseSegmentation.ts +6 -4
- package/src/nlp/Segmentation.ts +18 -13
- package/src/recognition/OpenAICloudSTT.ts +47 -29
- package/src/recognition/WhisperCppSTT.ts +26 -11
- package/src/recognition/WhisperSTT.ts +364 -49
- package/src/server/Client.ts +3 -2
- package/src/server/Worker.ts +3 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
- package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
- package/src/subtitles/Subtitles.ts +2 -2
- package/src/synthesis/GoogleTranslateTTS.ts +7 -21
- package/src/synthesis/VitsTTS.ts +31 -3
- package/src/tests/Test.ts +1 -38
- package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
- package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
- package/src/text-translation/DeepLTextTranslation.ts +88 -0
- package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
- package/src/text-translation/NLLBTextTranslation.ts +261 -21
- package/src/typings/Fillers.d.ts +25 -2
- package/src/utilities/BrowserRequestHeaders.ts +59 -0
- package/src/utilities/DynamicUint8Array.ts +39 -0
- package/src/utilities/Hashing.ts +14 -167
- package/src/utilities/LEB128.ts +273 -148
- package/src/utilities/LPVarInt.ts +292 -0
- package/src/utilities/Locale.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +1 -1
- package/src/utilities/PackageManager.ts +51 -30
- package/src/utilities/RandomGenerator.ts +12 -113
- package/src/utilities/Timeline.ts +162 -23
- package/src/utilities/Utilities.ts +40 -3
- package/src/utilities/VirtualFileReadStream.ts +109 -0
- package/src/utilities/WebReader.ts +9 -23
- package/dist/alignment/TextAlignment.js +0 -156
- package/dist/alignment/TextAlignment.js.map +0 -1
- package/dist/api/LanguageDetection.js.map +0 -1
- package/dist/api/Translation.js.map +0 -1
- package/src/alignment/TextAlignment.ts +0 -234
- /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
|
@@ -10,7 +10,7 @@ export class KWeightingFilter {
|
|
|
10
10
|
|
|
11
11
|
if (useStandard44100Filters) {
|
|
12
12
|
// These parameter values are taken from ITU-R BS.1770-2
|
|
13
|
-
// and designed only for a 44100 Hz
|
|
13
|
+
// and designed only for a 44100 Hz sample rate:
|
|
14
14
|
this.highShelfFilter = new BiquadFilter({
|
|
15
15
|
b0: 1.53512485958697,
|
|
16
16
|
b1: -2.69169618940638,
|
package/src/dsp/MelSpectogram.ts
CHANGED
|
@@ -2,7 +2,7 @@ import { RawAudio } from '../audio/AudioUtilities.js'
|
|
|
2
2
|
import { Logger } from '../utilities/Logger.js'
|
|
3
3
|
import * as FFT from './FFT.js'
|
|
4
4
|
|
|
5
|
-
export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbankCount: number, lowerFrequencyHz: number, upperFrequencyHz: number) {
|
|
5
|
+
export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbankCount: number, lowerFrequencyHz: number, upperFrequencyHz: number, windowType: FFT.WindowType = 'hann') {
|
|
6
6
|
const logger = new Logger()
|
|
7
7
|
|
|
8
8
|
logger.start('Compute mel filterbank')
|
|
@@ -18,15 +18,15 @@ export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number,
|
|
|
18
18
|
|
|
19
19
|
logger.end()
|
|
20
20
|
|
|
21
|
-
return computeMelSpectogramUsingFilterbanks(rawAudio, fftOrder, windowSize, hopLength, melFilterbanks)
|
|
21
|
+
return computeMelSpectogramUsingFilterbanks(rawAudio, fftOrder, windowSize, hopLength, melFilterbanks, windowType)
|
|
22
22
|
}
|
|
23
23
|
|
|
24
|
-
export async function computeMelSpectogramUsingFilterbanks(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbanks: Filterbank[]) {
|
|
24
|
+
export async function computeMelSpectogramUsingFilterbanks(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbanks: Filterbank[], windowType: FFT.WindowType = 'hann') {
|
|
25
25
|
const logger = new Logger()
|
|
26
26
|
|
|
27
27
|
logger.start('Compute short-time FFTs')
|
|
28
28
|
const audioSamples = rawAudio.audioChannels[0]
|
|
29
|
-
const fftFrames = await FFT.stftr(audioSamples, fftOrder, windowSize, hopLength,
|
|
29
|
+
const fftFrames = await FFT.stftr(audioSamples, fftOrder, windowSize, hopLength, windowType)
|
|
30
30
|
|
|
31
31
|
logger.start('Convert FFT frames to a mel spectogram')
|
|
32
32
|
const melSpectogram = fftFramesToMelSpectogram(fftFrames, filterbanks)
|
|
@@ -52,17 +52,26 @@ export function powerSpectrumToMelSpectrum(powerSpectrum: Float32Array, filterba
|
|
|
52
52
|
const filterbankStartIndex = filterbank.startIndex
|
|
53
53
|
const filterbankWeights = filterbank.weights
|
|
54
54
|
|
|
55
|
-
if (filterbankStartIndex
|
|
55
|
+
if (filterbankStartIndex === -1) {
|
|
56
56
|
continue
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
-
let
|
|
59
|
+
let melBandValue = 0
|
|
60
60
|
|
|
61
61
|
for (let i = 0; i < filterbankWeights.length; i++) {
|
|
62
|
-
|
|
62
|
+
const powerSpectrumIndex = filterbankStartIndex + i
|
|
63
|
+
|
|
64
|
+
if (powerSpectrumIndex >= powerSpectrum.length) {
|
|
65
|
+
break
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const weight = filterbankWeights[i]
|
|
69
|
+
const powerSpectrumValue = powerSpectrum[powerSpectrumIndex]
|
|
70
|
+
|
|
71
|
+
melBandValue += weight * powerSpectrumValue
|
|
63
72
|
}
|
|
64
73
|
|
|
65
|
-
melSpectrum[melBandIndex] =
|
|
74
|
+
melSpectrum[melBandIndex] = melBandValue
|
|
66
75
|
}
|
|
67
76
|
|
|
68
77
|
return melSpectrum
|
package/src/math/VectorMath.ts
CHANGED
|
@@ -177,7 +177,7 @@ export function scaleToSumTo1(vector: number[]) {
|
|
|
177
177
|
return scaledVector
|
|
178
178
|
}
|
|
179
179
|
|
|
180
|
-
export function normalizeVector(vector: number
|
|
180
|
+
export function normalizeVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population') {
|
|
181
181
|
if (vector.length == 0) {
|
|
182
182
|
throw new Error('Vector is empty')
|
|
183
183
|
}
|
|
@@ -329,15 +329,15 @@ export function varianceOfVectors(vectors: number[][], kind: 'population' | 'sam
|
|
|
329
329
|
return result
|
|
330
330
|
}
|
|
331
331
|
|
|
332
|
-
export function meanOfVector(vector: number
|
|
332
|
+
export function meanOfVector(vector: ArrayLike<number>) {
|
|
333
333
|
if (vector.length == 0) {
|
|
334
|
-
|
|
334
|
+
return 0
|
|
335
335
|
}
|
|
336
336
|
|
|
337
337
|
return sumVector(vector) / vector.length
|
|
338
338
|
}
|
|
339
339
|
|
|
340
|
-
export function medianOfVector(vector: number
|
|
340
|
+
export function medianOfVector(vector: ArrayLike<number>) {
|
|
341
341
|
if (vector.length == 0) {
|
|
342
342
|
throw new Error('Vector is empty')
|
|
343
343
|
}
|
|
@@ -345,13 +345,13 @@ export function medianOfVector(vector: number[]) {
|
|
|
345
345
|
return vector[Math.floor(vector.length / 2)]
|
|
346
346
|
}
|
|
347
347
|
|
|
348
|
-
export function stdDeviationOfVector(vector: number
|
|
348
|
+
export function stdDeviationOfVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population', mean?: number) {
|
|
349
349
|
return Math.sqrt(varianceOfVector(vector, kind, mean))
|
|
350
350
|
}
|
|
351
351
|
|
|
352
|
-
export function varianceOfVector(vector: number
|
|
352
|
+
export function varianceOfVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population', mean?: number) {
|
|
353
353
|
if (vector.length == 0) {
|
|
354
|
-
|
|
354
|
+
return 0
|
|
355
355
|
}
|
|
356
356
|
|
|
357
357
|
const sampleSizeMetric = kind == 'population' || vector.length == 1 ? vector.length : vector.length - 1
|
|
@@ -362,8 +362,8 @@ export function varianceOfVector(vector: number[], kind: 'population' | 'sample'
|
|
|
362
362
|
|
|
363
363
|
let result = 0.0
|
|
364
364
|
|
|
365
|
-
for (
|
|
366
|
-
result += (
|
|
365
|
+
for (let i = 0; i < vector.length; i++) {
|
|
366
|
+
result += (vector[i] - mean) ** 2
|
|
367
367
|
}
|
|
368
368
|
|
|
369
369
|
return result / sampleSizeMetric
|
|
@@ -456,11 +456,11 @@ export function meanSquaredError(actual: ArrayLike<number>, expected: ArrayLike<
|
|
|
456
456
|
return sum / featureCount
|
|
457
457
|
}
|
|
458
458
|
|
|
459
|
-
export function
|
|
460
|
-
return Math.sqrt(
|
|
459
|
+
export function euclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
460
|
+
return Math.sqrt(squaredEuclideanDistance(vector1, vector2))
|
|
461
461
|
}
|
|
462
462
|
|
|
463
|
-
export function
|
|
463
|
+
export function squaredEuclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
464
464
|
if (vector1.length !== vector2.length) {
|
|
465
465
|
throw new Error('Vectors are not the same length')
|
|
466
466
|
}
|
|
@@ -480,11 +480,11 @@ export function squaredEuclidianDistance(vector1: ArrayLike<number>, vector2: Ar
|
|
|
480
480
|
return sum
|
|
481
481
|
}
|
|
482
482
|
|
|
483
|
-
export function
|
|
484
|
-
return Math.sqrt(
|
|
483
|
+
export function euclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
484
|
+
return Math.sqrt(squaredEuclideanDistance13Dim(vector1, vector2))
|
|
485
485
|
}
|
|
486
486
|
|
|
487
|
-
export function
|
|
487
|
+
export function squaredEuclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
|
|
488
488
|
// Assumes the input has 13 dimensions (optimized for 13-dimensional MFCC vectors)
|
|
489
489
|
|
|
490
490
|
const result =
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
export async function splitChineseTextToWords_Jieba(text: string, fineGrained = false, useHMM = true) {
|
|
2
|
-
const jieba = await
|
|
2
|
+
const jieba = await getJiebaWasmInstance()
|
|
3
3
|
|
|
4
4
|
if (!fineGrained) {
|
|
5
5
|
return jieba.cut(text, useHMM)
|
|
@@ -58,10 +58,12 @@ export async function splitChineseTextToWords_Jieba(text: string, fineGrained =
|
|
|
58
58
|
}
|
|
59
59
|
|
|
60
60
|
let JiebaWasmInstance: typeof import('jieba-wasm')
|
|
61
|
-
|
|
61
|
+
|
|
62
|
+
async function getJiebaWasmInstance() {
|
|
62
63
|
if (!JiebaWasmInstance) {
|
|
63
|
-
const { default:
|
|
64
|
-
|
|
64
|
+
const { default: JiebaWasm } = await import('jieba-wasm')
|
|
65
|
+
|
|
66
|
+
JiebaWasmInstance = JiebaWasm
|
|
65
67
|
}
|
|
66
68
|
|
|
67
69
|
return JiebaWasmInstance
|
package/src/nlp/Segmentation.ts
CHANGED
|
@@ -11,8 +11,7 @@ const log = logToStderr
|
|
|
11
11
|
export const wordCharacterPattern = /[\p{Letter}\p{Number}]/u
|
|
12
12
|
export const punctuationPattern = /[\p{Punctuation}]/u
|
|
13
13
|
|
|
14
|
-
export const phraseSeparators = [',', ';', ':']
|
|
15
|
-
export const sentenceSeparators = ['.', '?', '!']
|
|
14
|
+
export const phraseSeparators = [',', ';', ':', ',', '、']
|
|
16
15
|
export const symbolWords = ['$', '€', '¢', '£', '¥', '©', '®', '™', '%', '&', '#', '~', '@', '+', '±', '÷', '/', '*', '=', '¼', '½', '¾']
|
|
17
16
|
|
|
18
17
|
export function isWordOrSymbolWord(str: string) {
|
|
@@ -226,29 +225,35 @@ export async function splitToWords(text: string, langCode: string): Promise<stri
|
|
|
226
225
|
}
|
|
227
226
|
}
|
|
228
227
|
|
|
229
|
-
export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType,
|
|
228
|
+
export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType, whitespaceProcessingMethod: WhitespaceProcessing) {
|
|
230
229
|
let paragraphs: string[] = []
|
|
231
230
|
|
|
232
|
-
if (paragraphBreaks
|
|
231
|
+
if (paragraphBreaks === 'single') {
|
|
233
232
|
paragraphs = text.split(/(\r?\n)+/g)
|
|
234
|
-
} else if (paragraphBreaks
|
|
233
|
+
} else if (paragraphBreaks === 'double') {
|
|
235
234
|
paragraphs = text.split(/(\r?\n)(\r?\n)+/g)
|
|
236
235
|
} else {
|
|
237
|
-
throw new Error(`Invalid paragraph break type: ${paragraphBreaks}`)
|
|
236
|
+
throw new Error(`Invalid paragraph break type: '${paragraphBreaks}'`)
|
|
238
237
|
}
|
|
239
238
|
|
|
240
|
-
|
|
241
|
-
paragraphs = paragraphs.map(p => p.replaceAll(/(\r?\n)+/g, ' '))
|
|
242
|
-
} else if (whitespace == 'collapse') {
|
|
243
|
-
paragraphs = paragraphs.map(p => p.replaceAll(/\s+/g, ' '))
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
paragraphs = paragraphs.map(p => p.trim())
|
|
239
|
+
paragraphs = paragraphs.map(p => applyWhitespaceProcessing(p.trim(), whitespaceProcessingMethod))
|
|
247
240
|
paragraphs = paragraphs.filter(p => p.length > 0)
|
|
248
241
|
|
|
249
242
|
return paragraphs
|
|
250
243
|
}
|
|
251
244
|
|
|
245
|
+
export function applyWhitespaceProcessing(text: string, whitespaceProcessingMethod: WhitespaceProcessing) {
|
|
246
|
+
if (whitespaceProcessingMethod === 'removeLineBreaks') {
|
|
247
|
+
return text.replaceAll(/(\r?\n)+/g, ' ')
|
|
248
|
+
} else if (whitespaceProcessingMethod === 'collapse') {
|
|
249
|
+
return text.replaceAll(/\s+/g, ' ')
|
|
250
|
+
} else if (whitespaceProcessingMethod === 'preserve') {
|
|
251
|
+
return text
|
|
252
|
+
} else {
|
|
253
|
+
throw new Error(`Invalid whitespace processing method: '${whitespaceProcessingMethod}'`)
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
|
|
252
257
|
export function splitToLines(text: string) {
|
|
253
258
|
return text.split(/\r?\n/g)
|
|
254
259
|
}
|
|
@@ -1,8 +1,10 @@
|
|
|
1
|
-
import { RawAudio } from '../audio/AudioUtilities.js';
|
|
2
1
|
import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
2
|
+
import { RawAudio } from '../audio/AudioUtilities.js'
|
|
3
|
+
import { createVirtualFileReadStreamForBuffer } from '../utilities/VirtualFileReadStream.js'
|
|
4
|
+
import { Logger } from '../utilities/Logger.js'
|
|
5
|
+
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
6
|
+
import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
|
|
7
|
+
import { alignSegments } from '../api/Alignment.js'
|
|
6
8
|
|
|
7
9
|
export async function recognize(rawAudio: RawAudio, languageCode: string, options: OpenAICloudSTTOptions, task: Task = 'transcribe') {
|
|
8
10
|
const logger = new Logger()
|
|
@@ -11,41 +13,56 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
|
|
|
11
13
|
|
|
12
14
|
options = extendDeep(defaultOpenAICloudSTTOptions, options)
|
|
13
15
|
|
|
16
|
+
if (options.requestWordTimestamps === undefined) {
|
|
17
|
+
options.requestWordTimestamps = options.baseURL === undefined
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
if (options.model === undefined) {
|
|
21
|
+
if (options.baseURL === undefined) {
|
|
22
|
+
options.model = 'whisper-1'
|
|
23
|
+
} else {
|
|
24
|
+
throw new Error(`A custom provider for the OpenAI Cloud API requires specifying a model name`)
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
|
|
14
28
|
const { default: OpenAI } = await import('openai')
|
|
15
29
|
const openai = new OpenAI(options)
|
|
16
30
|
|
|
17
31
|
logger.start('Encode audio to send')
|
|
18
32
|
const ffmpegOptions = FFMpegTranscoder.getDefaultFFMpegOptionsForSpeech('mp3')
|
|
19
33
|
const encodedAudio = await FFMpegTranscoder.encodeFromChannels(rawAudio, ffmpegOptions)
|
|
20
|
-
const
|
|
34
|
+
const virtualFileStream = createVirtualFileReadStreamForBuffer(encodedAudio, 'audio.mp3')
|
|
21
35
|
|
|
22
|
-
logger.start('
|
|
36
|
+
logger.start(options.baseURL ? `Send request to ${options.baseURL}` : 'Send request to OpenAI Cloud API')
|
|
23
37
|
|
|
24
38
|
let response: VerboseResponse
|
|
25
39
|
|
|
26
|
-
if (task =='transcribe') {
|
|
40
|
+
if (task == 'transcribe') {
|
|
41
|
+
const timestamp_granularities: ('word' | 'segment')[] | undefined =
|
|
42
|
+
options.requestWordTimestamps ? ['word', 'segment'] : undefined
|
|
43
|
+
|
|
27
44
|
response = await openai.audio.transcriptions.create({
|
|
28
|
-
file:
|
|
29
|
-
model: options.model
|
|
45
|
+
file: virtualFileStream,
|
|
46
|
+
model: options.model,
|
|
30
47
|
language: languageCode,
|
|
31
48
|
prompt: options.prompt,
|
|
32
49
|
response_format: 'verbose_json',
|
|
33
50
|
temperature: options.temperature,
|
|
34
|
-
timestamp_granularities
|
|
35
|
-
}) as VerboseResponse
|
|
51
|
+
timestamp_granularities,
|
|
52
|
+
}) as any as VerboseResponse
|
|
36
53
|
} else if (task == 'translate') {
|
|
37
54
|
response = await openai.audio.translations.create({
|
|
38
|
-
file:
|
|
39
|
-
model: options.model
|
|
55
|
+
file: virtualFileStream,
|
|
56
|
+
model: options.model,
|
|
40
57
|
prompt: options.prompt,
|
|
41
58
|
response_format: 'verbose_json',
|
|
42
59
|
temperature: options.temperature,
|
|
43
|
-
}) as VerboseResponse
|
|
60
|
+
}) as any as VerboseResponse
|
|
44
61
|
} else {
|
|
45
62
|
throw new Error(`Invalid task`)
|
|
46
63
|
}
|
|
47
64
|
|
|
48
|
-
const transcript = response.text
|
|
65
|
+
const transcript = response.text.trim()
|
|
49
66
|
|
|
50
67
|
let timeline: Timeline
|
|
51
68
|
|
|
@@ -57,12 +74,20 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
|
|
|
57
74
|
endTime: entry.end
|
|
58
75
|
}))
|
|
59
76
|
} else {
|
|
60
|
-
|
|
77
|
+
const segmentTimeline = response.segments.map<TimelineEntry>(entry => ({
|
|
61
78
|
type: 'segment',
|
|
62
79
|
text: entry.text,
|
|
63
80
|
startTime: entry.start,
|
|
64
81
|
endTime: entry.end
|
|
65
82
|
}))
|
|
83
|
+
|
|
84
|
+
if (task === 'transcribe') {
|
|
85
|
+
logger.start('Align segments')
|
|
86
|
+
|
|
87
|
+
timeline = await alignSegments(rawAudio, segmentTimeline, { language: languageCode })
|
|
88
|
+
} else {
|
|
89
|
+
timeline = segmentTimeline
|
|
90
|
+
}
|
|
66
91
|
}
|
|
67
92
|
|
|
68
93
|
logger.end()
|
|
@@ -70,17 +95,6 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
|
|
|
70
95
|
return { transcript, timeline }
|
|
71
96
|
}
|
|
72
97
|
|
|
73
|
-
class FileLikeBlob extends Blob {
|
|
74
|
-
constructor(
|
|
75
|
-
public readonly parts: BlobPart[],
|
|
76
|
-
public readonly name: string,
|
|
77
|
-
public readonly lastModified: number,
|
|
78
|
-
options: BlobPropertyBag,
|
|
79
|
-
) {
|
|
80
|
-
super(parts, options)
|
|
81
|
-
}
|
|
82
|
-
}
|
|
83
|
-
|
|
84
98
|
interface VerboseResponse {
|
|
85
99
|
task: string
|
|
86
100
|
language: string
|
|
@@ -115,7 +129,7 @@ interface VerboseResponse {
|
|
|
115
129
|
type Task = 'transcribe' | 'translate'
|
|
116
130
|
|
|
117
131
|
export interface OpenAICloudSTTOptions {
|
|
118
|
-
model?: 'whisper-1'
|
|
132
|
+
model?: 'whisper-1' | string
|
|
119
133
|
|
|
120
134
|
apiKey?: string
|
|
121
135
|
organization?: string
|
|
@@ -126,6 +140,8 @@ export interface OpenAICloudSTTOptions {
|
|
|
126
140
|
|
|
127
141
|
timeout?: number
|
|
128
142
|
maxRetries?: number
|
|
143
|
+
|
|
144
|
+
requestWordTimestamps?: boolean
|
|
129
145
|
}
|
|
130
146
|
|
|
131
147
|
export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
|
|
@@ -133,10 +149,12 @@ export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
|
|
|
133
149
|
organization: undefined,
|
|
134
150
|
baseURL: undefined,
|
|
135
151
|
|
|
136
|
-
model:
|
|
152
|
+
model: undefined,
|
|
137
153
|
temperature: 0,
|
|
138
154
|
prompt: undefined,
|
|
139
155
|
|
|
140
156
|
timeout: undefined,
|
|
141
157
|
maxRetries: 10,
|
|
158
|
+
|
|
159
|
+
requestWordTimestamps: undefined,
|
|
142
160
|
}
|
|
@@ -13,7 +13,7 @@ import { splitToLines } from '../nlp/Segmentation.js'
|
|
|
13
13
|
import { extendDeep } from '../utilities/ObjectUtilities.js'
|
|
14
14
|
import { formatLanguageCodeWithName, getShortLanguageCode } from '../utilities/Locale.js'
|
|
15
15
|
import { loadPackage } from '../utilities/PackageManager.js'
|
|
16
|
-
import { detectSpeechLanguageByParts } from '../api/
|
|
16
|
+
import { detectSpeechLanguageByParts } from '../api/SpeechLanguageDetection.js'
|
|
17
17
|
|
|
18
18
|
export async function recognize(
|
|
19
19
|
sourceRawAudio: RawAudio,
|
|
@@ -54,7 +54,7 @@ export async function recognize(
|
|
|
54
54
|
}
|
|
55
55
|
} else {
|
|
56
56
|
if (options.enableGPU) {
|
|
57
|
-
buildKind = 'cublas-
|
|
57
|
+
buildKind = 'cublas-12.4.0'
|
|
58
58
|
} else {
|
|
59
59
|
buildKind = 'cpu'
|
|
60
60
|
}
|
|
@@ -201,7 +201,7 @@ export async function recognize(
|
|
|
201
201
|
|
|
202
202
|
export async function detectLanguage(sourceRawAudio: RawAudio, modelName: WhisperModelName, modelPath: string) {
|
|
203
203
|
if (sourceRawAudio.sampleRate != 16000) {
|
|
204
|
-
throw new Error('Source audio must have a
|
|
204
|
+
throw new Error('Source audio must have a sample rate of 16000')
|
|
205
205
|
}
|
|
206
206
|
|
|
207
207
|
async function detectLanguageForPart(partAudio: RawAudio) {
|
|
@@ -240,6 +240,8 @@ async function parseResultObject(resultObject: WhisperCppVerboseResult, modelNam
|
|
|
240
240
|
|
|
241
241
|
let currentCorrectionTimeOffset = 0
|
|
242
242
|
|
|
243
|
+
let lastTokenEndOffset = 0
|
|
244
|
+
|
|
243
245
|
for (let segmentIndex = 0; segmentIndex < resultObject.transcription.length; segmentIndex++) {
|
|
244
246
|
const segmentObject = resultObject.transcription[segmentIndex]
|
|
245
247
|
|
|
@@ -248,6 +250,17 @@ async function parseResultObject(resultObject: WhisperCppVerboseResult, modelNam
|
|
|
248
250
|
for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex++) {
|
|
249
251
|
const tokenObject = tokens[tokenIndex]
|
|
250
252
|
|
|
253
|
+
// Workaround whisper.cpp issue with missing offsets by falling back to last known end offset
|
|
254
|
+
// when they are not included
|
|
255
|
+
if (!tokenObject.offsets) {
|
|
256
|
+
tokenObject.offsets = {
|
|
257
|
+
from: lastTokenEndOffset,
|
|
258
|
+
to: lastTokenEndOffset,
|
|
259
|
+
}
|
|
260
|
+
} else {
|
|
261
|
+
lastTokenEndOffset = tokenObject.offsets.to
|
|
262
|
+
}
|
|
263
|
+
|
|
251
264
|
if (tokenIndex === 0 && tokenObject.text === '[_BEG_]' && tokenObject.offsets.from === 0) {
|
|
252
265
|
currentCorrectionTimeOffset = segmentObject.offsets.from / 1000
|
|
253
266
|
}
|
|
@@ -370,7 +383,7 @@ export async function loadModelPackage(modelId: WhisperCppModelId | undefined, l
|
|
|
370
383
|
return { modelName, modelPath }
|
|
371
384
|
}
|
|
372
385
|
|
|
373
|
-
export type WhisperCppBuild = 'cpu' | 'cublas-
|
|
386
|
+
export type WhisperCppBuild = 'cpu' | 'cublas-12.4.0' | 'custom'
|
|
374
387
|
|
|
375
388
|
export async function loadExecutablePackage(buildKind: WhisperCppBuild) {
|
|
376
389
|
if (buildKind === 'custom') {
|
|
@@ -384,20 +397,20 @@ export async function loadExecutablePackage(buildKind: WhisperCppBuild) {
|
|
|
384
397
|
|
|
385
398
|
if (buildKind.startsWith('cublas-')) {
|
|
386
399
|
if (platform === 'win32' && arch === 'x64') {
|
|
387
|
-
packageName = `whisper.cpp-binaries-windows-x64-${buildKind}-latest
|
|
400
|
+
packageName = `whisper.cpp-binaries-windows-x64-${buildKind}-latest`
|
|
388
401
|
} else {
|
|
389
|
-
throw new Error(`GPU builds (NVIDIA CUDA only) are currently only available as packages for Windows x64. Please specify a custom path to
|
|
402
|
+
throw new Error(`whisper.cpp GPU builds (NVIDIA CUDA only) are currently only available as packages for Windows x64. Please specify a custom path to a whisper.cpp 'main' binary in the 'executablePath' option.`)
|
|
390
403
|
}
|
|
391
404
|
} else if (buildKind === 'cpu') {
|
|
392
405
|
if (platform === 'win32' && arch === 'x64') {
|
|
393
|
-
packageName = `whisper.cpp-binaries-windows-x64-cpu-latest
|
|
406
|
+
packageName = `whisper.cpp-binaries-windows-x64-cpu-latest`
|
|
394
407
|
} else if (platform === 'linux' && arch === 'x64') {
|
|
395
|
-
packageName = `whisper.cpp-binaries-linux-x64-cpu-latest
|
|
408
|
+
packageName = `whisper.cpp-binaries-linux-x64-cpu-latest`
|
|
396
409
|
} else {
|
|
397
|
-
throw new Error(`Couldn't find a matching whisper.cpp binary package. Please specify a custom path to
|
|
410
|
+
throw new Error(`Couldn't find a matching whisper.cpp binary package. Please specify a custom path to a whisper.cpp 'main' binary in the 'executablePath' option.`)
|
|
398
411
|
}
|
|
399
412
|
} else {
|
|
400
|
-
throw new Error(`
|
|
413
|
+
throw new Error(`Unsupported build kind '${buildKind}'`)
|
|
401
414
|
}
|
|
402
415
|
|
|
403
416
|
const packagePath = await loadPackage(packageName)
|
|
@@ -553,4 +566,6 @@ export type WhisperCppModelId =
|
|
|
553
566
|
'large-v2' |
|
|
554
567
|
'large-v2-q5_0' |
|
|
555
568
|
'large-v3' |
|
|
556
|
-
'large-v3-q5_0'
|
|
569
|
+
'large-v3-q5_0' |
|
|
570
|
+
`large-v3-turbo` |
|
|
571
|
+
`large-v3-turbo-q5_0`
|