echogarden 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -23
- package/data/schemas/options.json +177 -36
- package/dist/alignment/SpeechAlignment.d.ts +1 -1
- package/dist/alignment/SpeechAlignment.js +1 -1
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +1 -0
- package/dist/api/API.js +1 -0
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +1 -0
- package/dist/api/Alignment.d.ts +3 -3
- package/dist/api/Alignment.js +5 -10
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetection.d.ts +5 -7
- package/dist/api/LanguageDetection.js +3 -2
- package/dist/api/LanguageDetection.js.map +1 -1
- package/dist/api/Recognition.d.ts +4 -5
- package/dist/api/Recognition.js +5 -8
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/SourceSeparation.d.ts +2 -0
- package/dist/api/SourceSeparation.js +4 -2
- package/dist/api/SourceSeparation.js.map +1 -1
- package/dist/api/Synthesis.d.ts +3 -1
- package/dist/api/Synthesis.js +9 -10
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.d.ts +1 -1
- package/dist/api/Translation.js +4 -8
- package/dist/api/Translation.js.map +1 -1
- package/dist/api/TranslationAlignment.d.ts +31 -0
- package/dist/api/TranslationAlignment.js +121 -0
- package/dist/api/TranslationAlignment.js.map +1 -0
- package/dist/api/VoiceActivityDetection.d.ts +5 -1
- package/dist/api/VoiceActivityDetection.js +38 -2
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/audio/AudioPlayer.js +6 -1
- package/dist/audio/AudioPlayer.js.map +1 -1
- package/dist/cli/CLI.js +85 -0
- package/dist/cli/CLI.js.map +1 -1
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/math/MedianFilter.d.ts +5 -0
- package/dist/math/MedianFilter.js +102 -0
- package/dist/math/MedianFilter.js.map +1 -0
- package/dist/math/VectorMath.d.ts +0 -2
- package/dist/math/VectorMath.js +1 -25
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/recognition/OpenAICloudSTT.d.ts +1 -1
- package/dist/recognition/OpenAICloudSTT.js.map +1 -1
- package/dist/recognition/SileroSTT.d.ts +22 -1
- package/dist/recognition/SileroSTT.js +122 -95
- package/dist/recognition/SileroSTT.js.map +1 -1
- package/dist/recognition/WhisperCppSTT.js +1 -1
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.d.ts +52 -19
- package/dist/recognition/WhisperSTT.js +645 -494
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Server.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.d.ts +5 -3
- package/dist/source-separation/MDXNetSourceSeparation.js +26 -19
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +15 -9
- package/dist/speech-language-detection/SileroLanguageDetection.js +23 -16
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -0
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/synthesis/GoogleCloudTTS.js.map +1 -1
- package/dist/synthesis/VitsTTS.d.ts +8 -6
- package/dist/synthesis/VitsTTS.js +36 -31
- package/dist/synthesis/VitsTTS.js.map +1 -1
- package/dist/tests/Test.js.map +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +14 -0
- package/dist/utilities/OnnxUtilities.js +43 -0
- package/dist/utilities/OnnxUtilities.js.map +1 -0
- package/dist/utilities/Utilities.d.ts +4 -8
- package/dist/utilities/Utilities.js +35 -58
- package/dist/utilities/Utilities.js.map +1 -1
- package/dist/voice-activity-detection/SileroVAD.d.ts +5 -3
- package/dist/voice-activity-detection/SileroVAD.js +9 -11
- package/dist/voice-activity-detection/SileroVAD.js.map +1 -1
- package/docs/API.md +54 -34
- package/docs/CLI.md +25 -13
- package/docs/Contributing.md +4 -2
- package/docs/Engines.md +43 -32
- package/docs/Licenses.md +3 -4
- package/docs/Options.md +47 -11
- package/docs/Releases.md +4 -0
- package/docs/Server.md +8 -6
- package/docs/Tasklist.md +39 -52
- package/docs/Technical.md +1 -1
- package/package.json +8 -12
- package/src/alignment/SpeechAlignment.ts +1 -1
- package/src/api/API.ts +1 -0
- package/src/api/APIOptions.ts +1 -0
- package/src/api/Alignment.ts +10 -14
- package/src/api/LanguageDetection.ts +14 -10
- package/src/api/Recognition.ts +17 -10
- package/src/api/SourceSeparation.ts +7 -2
- package/src/api/Synthesis.ts +26 -11
- package/src/api/Translation.ts +14 -8
- package/src/api/TranslationAlignment.ts +213 -0
- package/src/api/VoiceActivityDetection.ts +66 -3
- package/src/audio/AudioPlayer.ts +6 -2
- package/src/cli/CLI.ts +121 -2
- package/src/dsp/FFT.ts +3 -0
- package/src/math/MedianFilter.ts +124 -0
- package/src/math/VectorMath.ts +1 -36
- package/src/recognition/OpenAICloudSTT.ts +27 -27
- package/src/recognition/SileroSTT.ts +149 -102
- package/src/recognition/WhisperCppSTT.ts +1 -1
- package/src/recognition/WhisperSTT.ts +961 -684
- package/src/server/Server.ts +1 -1
- package/src/source-separation/MDXNetSourceSeparation.ts +35 -19
- package/src/speech-language-detection/SileroLanguageDetection.ts +53 -33
- package/src/synthesis/EspeakTTS.ts +8 -0
- package/src/synthesis/GoogleCloudTTS.ts +12 -1
- package/src/synthesis/VitsTTS.ts +57 -46
- package/src/tests/Test.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +68 -0
- package/src/utilities/Utilities.ts +38 -66
- package/src/voice-activity-detection/SileroVAD.ts +15 -15
- package/dist/utilities/NdArrayUtilities.d.ts +0 -3
- package/dist/utilities/NdArrayUtilities.js +0 -23
- package/dist/utilities/NdArrayUtilities.js.map +0 -1
- package/src/utilities/NdArrayUtilities.ts +0 -31
package/src/math/VectorMath.ts
CHANGED
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
import createMedianFilter from 'moving-median'
|
|
2
|
-
|
|
3
1
|
export function covarianceMatrixOfSamples(samples: number[][], weights?: number[], biased = false) {
|
|
4
2
|
if (samples.length == 0) {
|
|
5
3
|
throw new Error('No vectors given')
|
|
@@ -507,7 +505,7 @@ export function cosineSimilarity(vector1: number[], vector2: number[]) {
|
|
|
507
505
|
squaredMagnitude2 += vector2[i] ** 2
|
|
508
506
|
}
|
|
509
507
|
|
|
510
|
-
const result = dotProduct / (Math.sqrt(squaredMagnitude1) * Math.sqrt(squaredMagnitude2))
|
|
508
|
+
const result = dotProduct / (Math.sqrt(squaredMagnitude1) * Math.sqrt(squaredMagnitude2) + 1e-40)
|
|
511
509
|
|
|
512
510
|
return zeroIfNaN(result)
|
|
513
511
|
}
|
|
@@ -651,28 +649,6 @@ export function indexOfMin(vector: number[]) {
|
|
|
651
649
|
return result
|
|
652
650
|
}
|
|
653
651
|
|
|
654
|
-
export function exponentialSmoothingMeanOfVectors(vectors: number[][], smoothingFactor: number) {
|
|
655
|
-
const vectorCount = vectors.length
|
|
656
|
-
|
|
657
|
-
if (vectorCount == 0) {
|
|
658
|
-
return []
|
|
659
|
-
}
|
|
660
|
-
|
|
661
|
-
const featureCount = vectors[0].length
|
|
662
|
-
|
|
663
|
-
const currentEstimate = createVector(featureCount)
|
|
664
|
-
|
|
665
|
-
for (let i = 0; i < vectorCount; i++) {
|
|
666
|
-
for (let j = 0; j < featureCount; j++) {
|
|
667
|
-
const value = vectors[i][j]
|
|
668
|
-
|
|
669
|
-
currentEstimate[j] = (smoothingFactor * value) + ((1 - smoothingFactor) * currentEstimate[j])
|
|
670
|
-
}
|
|
671
|
-
}
|
|
672
|
-
|
|
673
|
-
return currentEstimate
|
|
674
|
-
}
|
|
675
|
-
|
|
676
652
|
export function sigmoid(x: number) {
|
|
677
653
|
const result = 1 / (1 + Math.exp(-x))
|
|
678
654
|
|
|
@@ -728,17 +704,6 @@ export function hammingDistance(value1: number, value2: number, bitLength = 32)
|
|
|
728
704
|
return result
|
|
729
705
|
}
|
|
730
706
|
|
|
731
|
-
export function medianFilter(vector: number[], width: number) {
|
|
732
|
-
const filter = createMedianFilter(width)
|
|
733
|
-
const result = []
|
|
734
|
-
|
|
735
|
-
for (let i = 0; i < vector.length; i++) {
|
|
736
|
-
result.push(filter(vector[i]))
|
|
737
|
-
}
|
|
738
|
-
|
|
739
|
-
return result
|
|
740
|
-
}
|
|
741
|
-
|
|
742
707
|
export function createVectorArray(vectorCount: number, featureCount: number, initialValue = 0.0) {
|
|
743
708
|
const result: number[][] = new Array(vectorCount)
|
|
744
709
|
|
|
@@ -81,33 +81,6 @@ class FileLikeBlob extends Blob {
|
|
|
81
81
|
}
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
-
export interface OpenAICloudSTTOptions {
|
|
85
|
-
model?: 'whisper-1'
|
|
86
|
-
|
|
87
|
-
apiKey?: string
|
|
88
|
-
organization?: string
|
|
89
|
-
baseURL?: string
|
|
90
|
-
|
|
91
|
-
temperature?: number
|
|
92
|
-
prompt?: string
|
|
93
|
-
|
|
94
|
-
timeout?: number
|
|
95
|
-
maxRetries?: number
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
|
|
99
|
-
apiKey: undefined,
|
|
100
|
-
organization: undefined,
|
|
101
|
-
baseURL: undefined,
|
|
102
|
-
|
|
103
|
-
model: 'whisper-1',
|
|
104
|
-
temperature: 0,
|
|
105
|
-
prompt: undefined,
|
|
106
|
-
|
|
107
|
-
timeout: undefined,
|
|
108
|
-
maxRetries: 10,
|
|
109
|
-
}
|
|
110
|
-
|
|
111
84
|
interface VerboseResponse {
|
|
112
85
|
task: string
|
|
113
86
|
language: string
|
|
@@ -140,3 +113,30 @@ interface VerboseResponse {
|
|
|
140
113
|
}
|
|
141
114
|
|
|
142
115
|
type Task = 'transcribe' | 'translate'
|
|
116
|
+
|
|
117
|
+
export interface OpenAICloudSTTOptions {
|
|
118
|
+
model?: 'whisper-1'
|
|
119
|
+
|
|
120
|
+
apiKey?: string
|
|
121
|
+
organization?: string
|
|
122
|
+
baseURL?: string
|
|
123
|
+
|
|
124
|
+
temperature?: number
|
|
125
|
+
prompt?: string
|
|
126
|
+
|
|
127
|
+
timeout?: number
|
|
128
|
+
maxRetries?: number
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
|
|
132
|
+
apiKey: undefined,
|
|
133
|
+
organization: undefined,
|
|
134
|
+
baseURL: undefined,
|
|
135
|
+
|
|
136
|
+
model: 'whisper-1',
|
|
137
|
+
temperature: 0,
|
|
138
|
+
prompt: undefined,
|
|
139
|
+
|
|
140
|
+
timeout: undefined,
|
|
141
|
+
maxRetries: 10,
|
|
142
|
+
}
|
|
@@ -4,167 +4,204 @@ import { Logger } from '../utilities/Logger.js'
|
|
|
4
4
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
5
5
|
import { Timeline } from '../utilities/Timeline.js'
|
|
6
6
|
import { RawAudio, getRawAudioDuration } from '../audio/AudioUtilities.js'
|
|
7
|
-
import { readAndParseJsonFile
|
|
8
|
-
import type * as Onnx from 'onnxruntime-node'
|
|
7
|
+
import { readAndParseJsonFile } from '../utilities/FileSystem.js'
|
|
9
8
|
import path from 'path'
|
|
10
9
|
|
|
10
|
+
import type * as Onnx from 'onnxruntime-node'
|
|
11
|
+
import { OnnxExecutionProvider, getOnnxSessionOptions } from '../utilities/OnnxUtilities.js'
|
|
12
|
+
|
|
11
13
|
const log = logToStderr
|
|
12
14
|
|
|
13
|
-
export async function recognize(
|
|
14
|
-
|
|
15
|
-
|
|
15
|
+
export async function recognize(
|
|
16
|
+
rawAudio: RawAudio,
|
|
17
|
+
modelDirectoryPath: string,
|
|
18
|
+
executionProviders: OnnxExecutionProvider[]) {
|
|
16
19
|
|
|
17
|
-
const
|
|
20
|
+
const silero = new SileroSTT(modelDirectoryPath, executionProviders)
|
|
18
21
|
|
|
19
|
-
const
|
|
20
|
-
const labelsPath = path.join(modelDirectory, 'labels.json')
|
|
22
|
+
const result = await silero.recognize(rawAudio)
|
|
21
23
|
|
|
22
|
-
|
|
24
|
+
return result
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export class SileroSTT {
|
|
28
|
+
session?: Onnx.InferenceSession
|
|
29
|
+
labels?: string[]
|
|
23
30
|
|
|
24
|
-
|
|
25
|
-
|
|
31
|
+
constructor(
|
|
32
|
+
public readonly modelDirectoryPath: string,
|
|
33
|
+
public readonly executionProviders: OnnxExecutionProvider[]
|
|
34
|
+
) {
|
|
26
35
|
}
|
|
27
36
|
|
|
28
|
-
|
|
37
|
+
async recognize(rawAudio: RawAudio) {
|
|
38
|
+
const logger = new Logger()
|
|
29
39
|
|
|
30
|
-
|
|
40
|
+
await this.initializeIfNeeded()
|
|
31
41
|
|
|
32
|
-
|
|
42
|
+
logger.start('Recognize with silero model')
|
|
33
43
|
|
|
34
|
-
|
|
44
|
+
const audioSamples = rawAudio.audioChannels[0]
|
|
35
45
|
|
|
36
|
-
|
|
46
|
+
const Onnx = await import('onnxruntime-node')
|
|
37
47
|
|
|
38
|
-
|
|
48
|
+
const inputTensor = new Onnx.Tensor('float32', audioSamples, [1, audioSamples.length])
|
|
49
|
+
const inputs = { input: inputTensor }
|
|
39
50
|
|
|
40
|
-
|
|
51
|
+
const results = await this.session!.run(inputs)
|
|
41
52
|
|
|
42
|
-
|
|
53
|
+
const rawResultValues = results['output'].data as Float32Array
|
|
43
54
|
|
|
44
|
-
|
|
55
|
+
const labels = this.labels!
|
|
45
56
|
|
|
46
|
-
|
|
47
|
-
tokenResults.push(rawResultValues.subarray(i, i + labels.length))
|
|
48
|
-
}
|
|
57
|
+
const tokenResults: Float32Array[] = []
|
|
49
58
|
|
|
50
|
-
|
|
59
|
+
for (let i = 0; i < rawResultValues.length; i += labels.length) {
|
|
60
|
+
tokenResults.push(rawResultValues.subarray(i, i + labels.length))
|
|
61
|
+
}
|
|
51
62
|
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
63
|
+
const tokens: string[] = []
|
|
64
|
+
|
|
65
|
+
for (const tokenResult of tokenResults) {
|
|
66
|
+
const bestCandidateIndex = indexOfMax(new Array(...tokenResult))
|
|
67
|
+
tokens.push(labels[bestCandidateIndex])
|
|
68
|
+
}
|
|
56
69
|
|
|
57
|
-
|
|
70
|
+
//log(tokens.join('|'))
|
|
58
71
|
|
|
59
|
-
|
|
72
|
+
const result = this.tokensToTimeline(tokens, getRawAudioDuration(rawAudio))
|
|
60
73
|
|
|
61
|
-
|
|
74
|
+
logger.end()
|
|
62
75
|
|
|
63
|
-
|
|
64
|
-
}
|
|
76
|
+
return result
|
|
77
|
+
}
|
|
65
78
|
|
|
66
|
-
|
|
67
|
-
|
|
79
|
+
private async initializeIfNeeded() {
|
|
80
|
+
if (this.session) {
|
|
81
|
+
return
|
|
82
|
+
}
|
|
68
83
|
|
|
69
|
-
|
|
70
|
-
let tokenGroupIndexes: number[][] = [[]]
|
|
84
|
+
const logger = new Logger()
|
|
71
85
|
|
|
72
|
-
|
|
73
|
-
const token = tokens[i]
|
|
86
|
+
logger.start('Create ONNX inference session')
|
|
74
87
|
|
|
75
|
-
|
|
76
|
-
if (decodedTokens.length > 0) {
|
|
77
|
-
const previousDecodedToken = decodedTokens[decodedTokens.length - 1]
|
|
78
|
-
decodedTokens.push('$')
|
|
79
|
-
decodedTokens.push(previousDecodedToken)
|
|
88
|
+
const Onnx = await import('onnxruntime-node')
|
|
80
89
|
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
decodedTokens.push(' ')
|
|
84
|
-
tokenGroupIndexes.push([])
|
|
85
|
-
}
|
|
90
|
+
const modelPath = path.join(this.modelDirectoryPath, 'model.onnx')
|
|
91
|
+
const labelsPath = path.join(this.modelDirectoryPath, 'labels.json')
|
|
86
92
|
|
|
87
|
-
|
|
88
|
-
}
|
|
93
|
+
this.labels = await readAndParseJsonFile(labelsPath)
|
|
89
94
|
|
|
90
|
-
|
|
91
|
-
continue
|
|
92
|
-
}
|
|
95
|
+
const onnxSessionOptions = getOnnxSessionOptions({ executionProviders: this.executionProviders })
|
|
93
96
|
|
|
94
|
-
|
|
97
|
+
this.session = await Onnx.InferenceSession.create(modelPath, onnxSessionOptions)
|
|
95
98
|
|
|
96
|
-
|
|
97
|
-
tokenGroupIndexes.push([])
|
|
98
|
-
} else {
|
|
99
|
-
tokenGroupIndexes[tokenGroupIndexes.length - 1].push(i)
|
|
100
|
-
}
|
|
99
|
+
logger.end()
|
|
101
100
|
}
|
|
102
101
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
for (let i = 0; i < decodedTokens.length; i++) {
|
|
106
|
-
const currentToken = decodedTokens[i]
|
|
107
|
-
const previousToken = decodedTokens[i - 1]
|
|
102
|
+
private tokensToTimeline(tokens: string[], totalDuration: number) {
|
|
103
|
+
const tokenCount = tokens.length
|
|
108
104
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
}
|
|
112
|
-
}
|
|
105
|
+
const decodedTokens: string[] = []
|
|
106
|
+
let tokenGroupIndexes: number[][] = [[]]
|
|
113
107
|
|
|
114
|
-
|
|
108
|
+
for (let i = 0; i < tokenCount; i++) {
|
|
109
|
+
const token = tokens[i]
|
|
115
110
|
|
|
116
|
-
|
|
111
|
+
if (token == '2') {
|
|
112
|
+
if (decodedTokens.length > 0) {
|
|
113
|
+
const previousDecodedToken = decodedTokens[decodedTokens.length - 1]
|
|
114
|
+
decodedTokens.push('$')
|
|
115
|
+
decodedTokens.push(previousDecodedToken)
|
|
117
116
|
|
|
118
|
-
|
|
119
|
-
|
|
117
|
+
tokenGroupIndexes[tokenGroupIndexes.length - 1].push(i)
|
|
118
|
+
} else {
|
|
119
|
+
decodedTokens.push(' ')
|
|
120
|
+
tokenGroupIndexes.push([])
|
|
121
|
+
}
|
|
120
122
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
+
continue
|
|
124
|
+
}
|
|
123
125
|
|
|
124
|
-
if (
|
|
125
|
-
|
|
126
|
+
if (token == '_') {
|
|
127
|
+
continue
|
|
126
128
|
}
|
|
127
129
|
|
|
128
|
-
|
|
130
|
+
decodedTokens.push(token)
|
|
129
131
|
|
|
130
|
-
if (
|
|
131
|
-
|
|
132
|
+
if (token == ' ') {
|
|
133
|
+
tokenGroupIndexes.push([])
|
|
132
134
|
} else {
|
|
133
|
-
|
|
135
|
+
tokenGroupIndexes[tokenGroupIndexes.length - 1].push(i)
|
|
134
136
|
}
|
|
137
|
+
}
|
|
135
138
|
|
|
136
|
-
|
|
139
|
+
let decodedString = ''
|
|
140
|
+
|
|
141
|
+
for (let i = 0; i < decodedTokens.length; i++) {
|
|
142
|
+
const currentToken = decodedTokens[i]
|
|
143
|
+
const previousToken = decodedTokens[i - 1]
|
|
144
|
+
|
|
145
|
+
if (currentToken != '$' && (previousToken != currentToken || previousToken == undefined)) {
|
|
146
|
+
decodedString += currentToken
|
|
147
|
+
}
|
|
137
148
|
}
|
|
138
|
-
}
|
|
139
149
|
|
|
140
|
-
|
|
150
|
+
decodedString = decodedString.trim()
|
|
151
|
+
|
|
152
|
+
tokenGroupIndexes = tokenGroupIndexes.filter(group => group.length > 0)
|
|
153
|
+
|
|
154
|
+
if (tokenGroupIndexes.length > 0) {
|
|
155
|
+
let currentCorrection = Math.min(tokenGroupIndexes[0][0], 1.5)
|
|
141
156
|
|
|
142
|
-
|
|
157
|
+
for (let i = 0; i < tokenGroupIndexes.length; i++) {
|
|
158
|
+
const group = tokenGroupIndexes[i]
|
|
143
159
|
|
|
144
|
-
|
|
160
|
+
if (group.length == 1) {
|
|
161
|
+
group.push(group[0])
|
|
162
|
+
}
|
|
145
163
|
|
|
146
|
-
|
|
147
|
-
const text = words[i]
|
|
164
|
+
group[0] -= currentCorrection
|
|
148
165
|
|
|
149
|
-
|
|
150
|
-
|
|
166
|
+
if (i == tokenGroupIndexes.length - 1) {
|
|
167
|
+
currentCorrection = Math.min(tokenCount - i, 1.5)
|
|
168
|
+
} else {
|
|
169
|
+
currentCorrection = Math.min((tokenGroupIndexes[i + 1][0] - group[group.length - 1]) / 2, 1.5)
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
group[group.length - 1] += currentCorrection
|
|
173
|
+
}
|
|
151
174
|
}
|
|
152
175
|
|
|
153
|
-
const
|
|
154
|
-
const startTime = group[0] * timeMultiplier
|
|
155
|
-
const endTime = group[group.length - 1] * timeMultiplier
|
|
176
|
+
const words = decodedString.split(' ')
|
|
156
177
|
|
|
157
|
-
|
|
158
|
-
type: 'word',
|
|
159
|
-
text: text,
|
|
160
|
-
startTime,
|
|
161
|
-
endTime,
|
|
162
|
-
})
|
|
163
|
-
}
|
|
178
|
+
const timeMultiplier = totalDuration / tokenCount
|
|
164
179
|
|
|
165
|
-
|
|
180
|
+
const timeline: Timeline = []
|
|
166
181
|
|
|
167
|
-
|
|
182
|
+
for (let i = 0; i < words.length; i++) {
|
|
183
|
+
const text = words[i]
|
|
184
|
+
|
|
185
|
+
if (!wordCharacterPattern.test(text)) {
|
|
186
|
+
continue
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const group = tokenGroupIndexes[i]
|
|
190
|
+
const startTime = group[0] * timeMultiplier
|
|
191
|
+
const endTime = group[group.length - 1] * timeMultiplier
|
|
192
|
+
|
|
193
|
+
timeline.push({
|
|
194
|
+
type: 'word',
|
|
195
|
+
text: text,
|
|
196
|
+
startTime,
|
|
197
|
+
endTime,
|
|
198
|
+
})
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
timeline[timeline.length - 1].endTime = totalDuration
|
|
202
|
+
|
|
203
|
+
return { transcript: decodedString, timeline }
|
|
204
|
+
}
|
|
168
205
|
}
|
|
169
206
|
|
|
170
207
|
export const languageCodeToPackageName: { [languageCode: string]: string } = {
|
|
@@ -173,3 +210,13 @@ export const languageCodeToPackageName: { [languageCode: string]: string } = {
|
|
|
173
210
|
'de': 'silero-de-v1',
|
|
174
211
|
'uk': 'silero-ua-v3',
|
|
175
212
|
}
|
|
213
|
+
|
|
214
|
+
export interface SileroRecognitionOptions {
|
|
215
|
+
modelPath?: string
|
|
216
|
+
provider?: OnnxExecutionProvider
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
export const defaultSileroRecognitionOptions: SileroRecognitionOptions = {
|
|
220
|
+
modelPath: undefined,
|
|
221
|
+
provider: undefined,
|
|
222
|
+
}
|
|
@@ -233,7 +233,7 @@ export async function detectLanguage(sourceRawAudio: RawAudio, modelName: Whispe
|
|
|
233
233
|
async function parseResultObject(resultObject: WhisperCppVerboseResult, modelName: WhisperModelName, totalDuration: number, enableDTW: boolean): Promise<RecognitionResult> {
|
|
234
234
|
const { Whisper } = await import('../recognition/WhisperSTT.js')
|
|
235
235
|
|
|
236
|
-
const whisper = new Whisper(modelName, '')
|
|
236
|
+
const whisper = new Whisper(modelName, '', [], [])
|
|
237
237
|
await whisper.initializeTokenizerIfNeeded()
|
|
238
238
|
|
|
239
239
|
const tokenTimeline: Timeline = []
|