echogarden 1.4.3 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/data/schemas/options.json +267 -19
- package/data/tables/lcid-table.json +9 -0
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
- package/dist/alignment/DTWMfccSequenceAlignment.js +9 -5
- package/dist/alignment/DTWMfccSequenceAlignment.js.map +1 -1
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +4 -4
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
- package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +10 -1
- package/dist/alignment/SemanticTextAlignment.js +336 -0
- package/dist/alignment/SemanticTextAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +2 -1
- package/dist/alignment/SpeechAlignment.js +106 -38
- package/dist/alignment/SpeechAlignment.js.map +1 -1
- package/dist/api/API.d.ts +7 -2
- package/dist/api/API.js +7 -2
- package/dist/api/API.js.map +1 -1
- package/dist/api/APIOptions.d.ts +4 -1
- package/dist/api/Alignment.d.ts +1 -1
- package/dist/api/Alignment.js +14 -6
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/LanguageDetectionCommon.d.ts +6 -0
- package/dist/api/LanguageDetectionCommon.js +2 -0
- package/dist/api/LanguageDetectionCommon.js.map +1 -0
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
- package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
- package/dist/api/SpeechLanguageDetection.js.map +1 -0
- package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
- package/dist/api/SpeechTranslation.js.map +1 -0
- package/dist/api/Synthesis.js +4 -4
- package/dist/api/TextLanguageDetection.d.ts +21 -0
- package/dist/api/TextLanguageDetection.js +67 -0
- package/dist/api/TextLanguageDetection.js.map +1 -0
- package/dist/api/TextTranslation.d.ts +5 -0
- package/dist/api/TextTranslation.js +4 -0
- package/dist/api/TextTranslation.js.map +1 -0
- package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
- package/dist/api/TimelineTranslationAlignment.js +92 -0
- package/dist/api/TimelineTranslationAlignment.js.map +1 -0
- package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
- package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
- package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
- package/dist/api/TranslationAlignment.d.ts +4 -3
- package/dist/api/TranslationAlignment.js +9 -8
- package/dist/api/TranslationAlignment.js.map +1 -1
- package/dist/api/VoiceActivityDetection.js +16 -1
- package/dist/api/VoiceActivityDetection.js.map +1 -1
- package/dist/cli/CLI.d.ts +27 -7
- package/dist/cli/CLI.js +205 -34
- package/dist/cli/CLI.js.map +1 -1
- package/dist/codecs/FFMpegTranscoder.js +7 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -1
- package/dist/dsp/FFT.d.ts +1 -1
- package/dist/dsp/FFT.js +6 -0
- package/dist/dsp/FFT.js.map +1 -1
- package/dist/dsp/KWeightingFilter.js +1 -1
- package/dist/dsp/KWeightingFilter.js.map +1 -1
- package/dist/dsp/MelSpectogram.d.ts +3 -2
- package/dist/dsp/MelSpectogram.js +14 -8
- package/dist/dsp/MelSpectogram.js.map +1 -1
- package/dist/math/VectorMath.d.ts +22 -20
- package/dist/math/VectorMath.js +57 -30
- package/dist/math/VectorMath.js.map +1 -1
- package/dist/recognition/WhisperCppSTT.d.ts +1 -1
- package/dist/recognition/WhisperCppSTT.js +2 -2
- package/dist/recognition/WhisperCppSTT.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +9 -6
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/server/Client.d.ts +3 -2
- package/dist/server/Client.js.map +1 -1
- package/dist/server/Worker.d.ts +3 -2
- package/dist/server/Worker.js +3 -2
- package/dist/server/Worker.js.map +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
- package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +13 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
- package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
- package/dist/synthesis/EspeakTTS.js +4 -5
- package/dist/synthesis/EspeakTTS.js.map +1 -1
- package/dist/tests/Test.js +0 -8
- package/dist/tests/Test.js.map +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
- package/dist/text-translation/NLLBTextTranslation.js +1 -1
- package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
- package/dist/utilities/Locale.d.ts +1 -1
- package/dist/utilities/Locale.js +1 -1
- package/dist/utilities/OnnxUtilities.d.ts +1 -1
- package/dist/utilities/PackageManager.js +8 -2
- package/dist/utilities/PackageManager.js.map +1 -1
- package/dist/utilities/Timeline.d.ts +1 -0
- package/dist/utilities/Timeline.js +12 -0
- package/dist/utilities/Timeline.js.map +1 -1
- package/docs/API.md +81 -3
- package/docs/CLI.md +51 -1
- package/docs/Engines.md +24 -0
- package/docs/Options.md +33 -1
- package/docs/Tasklist.md +4 -1
- package/package.json +11 -11
- package/src/alignment/DTWMfccSequenceAlignment.ts +11 -5
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +4 -4
- package/src/alignment/SemanticTextAlignment.ts +467 -0
- package/src/alignment/SpeechAlignment.ts +180 -53
- package/src/api/API.ts +18 -2
- package/src/api/APIOptions.ts +14 -1
- package/src/api/Alignment.ts +32 -10
- package/src/api/LanguageDetectionCommon.ts +7 -0
- package/src/api/Recognition.ts +2 -0
- package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
- package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
- package/src/api/Synthesis.ts +4 -4
- package/src/api/TextLanguageDetection.ts +116 -0
- package/src/api/TextTranslation.ts +9 -0
- package/src/api/TimelineTranslationAlignment.ts +162 -0
- package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
- package/src/api/TranslationAlignment.ts +12 -10
- package/src/api/VoiceActivityDetection.ts +24 -3
- package/src/cli/CLI.ts +276 -34
- package/src/codecs/FFMpegTranscoder.ts +6 -0
- package/src/dsp/FFT.ts +8 -2
- package/src/dsp/KWeightingFilter.ts +1 -1
- package/src/dsp/MelSpectogram.ts +17 -8
- package/src/math/VectorMath.ts +87 -52
- package/src/recognition/WhisperCppSTT.ts +2 -2
- package/src/recognition/WhisperSTT.ts +10 -6
- package/src/server/Client.ts +3 -2
- package/src/server/Worker.ts +3 -2
- package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
- package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
- package/src/synthesis/EspeakTTS.ts +6 -8
- package/src/tests/Test.ts +1 -12
- package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
- package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
- package/src/text-translation/NLLBTextTranslation.ts +1 -1
- package/src/utilities/Locale.ts +1 -1
- package/src/utilities/OnnxUtilities.ts +1 -1
- package/src/utilities/PackageManager.ts +9 -2
- package/src/utilities/Timeline.ts +14 -0
- package/dist/alignment/TextAlignment.js +0 -63
- package/dist/alignment/TextAlignment.js.map +0 -1
- package/dist/api/LanguageDetection.js.map +0 -1
- package/dist/api/Translation.js.map +0 -1
- package/src/alignment/TextAlignment.ts +0 -96
- /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
package/docs/Options.md
CHANGED
|
@@ -271,7 +271,8 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
|
|
|
271
271
|
|
|
272
272
|
**General**:
|
|
273
273
|
* `engine`: alignment algorithm to use, can only be `whisper`. Defaults to `whisper`
|
|
274
|
-
* `
|
|
274
|
+
* `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
|
|
275
|
+
* `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
|
|
275
276
|
* `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
|
|
276
277
|
* `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
|
|
277
278
|
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
@@ -284,6 +285,37 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
|
|
|
284
285
|
* `whisper.encoderProvider`: encoder ONNX execution provider. See details in recognition section above
|
|
285
286
|
* `whisper.decoderProvider`: decoder ONNX execution provider. See details in recognition section above
|
|
286
287
|
|
|
288
|
+
## Speech-to-transcript-and-translation alignment
|
|
289
|
+
|
|
290
|
+
Applies to CLI operation: `align-transcript-and-translation`, API method: `alignTranscriptAndTranslation`
|
|
291
|
+
|
|
292
|
+
**General**:
|
|
293
|
+
* `engine`: can only be `two-stage`. Defaults to `two-stage`
|
|
294
|
+
* `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
|
|
295
|
+
* `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
|
|
296
|
+
* `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
|
|
297
|
+
* `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
|
|
298
|
+
* `alignment`: prefix to provide options for alignment. Options detailed in section for alignment
|
|
299
|
+
* `timelineAlignment`: prefix to provide options for timeline alignment. Options detailed in section for timeline alignment
|
|
300
|
+
* `vad`: prefix to provide options for voice activity detection when `crop` is set to `true`. Options detailed in section for voice activity detection
|
|
301
|
+
* `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
|
|
302
|
+
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
303
|
+
|
|
304
|
+
## Timeline-to-translated-text alignment
|
|
305
|
+
|
|
306
|
+
Applies to CLI operation: `align-timeline-translation`, API method: `alignTimelineTranslation`
|
|
307
|
+
|
|
308
|
+
**General**:
|
|
309
|
+
* `engine`: alignment engine to use. Can only be `e5`. Defaults to `e5`
|
|
310
|
+
* `sourceLanguage`: language code for the source timeline. Auto-detected from timeline if not set
|
|
311
|
+
* `targetLanguage`: language code for the translated transcript. Auto-detected if not set
|
|
312
|
+
* `audio`: spoken audio to play when previewing the result in the CLI (not required or used by the alignment itself). Optional
|
|
313
|
+
* `languageDetection`: prefix to provide options for language detection. Options detailed in section for text language detection
|
|
314
|
+
* `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
|
|
315
|
+
|
|
316
|
+
**E5**:
|
|
317
|
+
* `e5.model`: E5 model to use. Defaults to `e5-small-fp16` (support for additional models will be added in the future)
|
|
318
|
+
|
|
287
319
|
## Language detection
|
|
288
320
|
|
|
289
321
|
### Speech language detection
|
package/docs/Tasklist.md
CHANGED
|
@@ -5,7 +5,6 @@
|
|
|
5
5
|
### Alignment / DTW-RA
|
|
6
6
|
|
|
7
7
|
* In DTW-RA, a recognized transcript including something like "Question 2.What does Juan", where "2.What" has a point in the middle, is breaking playback of the timeline
|
|
8
|
-
* DTW-RA will not work correctly with Polish language texts, due to issues with the eSpeak engine pronouncing `|` characters, which are intended to be used as separators and ignored by all other eSpeak languages
|
|
9
8
|
|
|
10
9
|
### Synthesis
|
|
11
10
|
|
|
@@ -159,6 +158,10 @@
|
|
|
159
158
|
|
|
160
159
|
### Alignment
|
|
161
160
|
|
|
161
|
+
### Alignment / DTW
|
|
162
|
+
* Accept percentages like `20%` in the `windowDuration` option
|
|
163
|
+
* For the `granularity` option, add more granularities like `xxx-low` and `xxxx-low` (should the naming be changed? Maybe transition to a new naming scheme?)
|
|
164
|
+
|
|
162
165
|
### Alignment / DTW-RA
|
|
163
166
|
|
|
164
167
|
### Alignment / Whisper
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "echogarden",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.5.0",
|
|
4
4
|
"description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
|
|
5
5
|
"author": "Rotem Dan",
|
|
6
6
|
"license": "GPL-3.0",
|
|
@@ -55,8 +55,8 @@
|
|
|
55
55
|
"echogarden": "./dist/cli/CLILauncher.js"
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
|
-
"@aws-sdk/client-polly": "^3.
|
|
59
|
-
"@aws-sdk/client-transcribe-streaming": "^3.
|
|
58
|
+
"@aws-sdk/client-polly": "^3.583.0",
|
|
59
|
+
"@aws-sdk/client-transcribe-streaming": "^3.583.0",
|
|
60
60
|
"@echogarden/espeak-ng-emscripten": "^0.1.2",
|
|
61
61
|
"@echogarden/fasttext-wasm": "^0.1.0",
|
|
62
62
|
"@echogarden/flite-wasi": "^0.1.1",
|
|
@@ -76,24 +76,24 @@
|
|
|
76
76
|
"command-exists": "^1.2.9",
|
|
77
77
|
"compromise": "^14.13.0",
|
|
78
78
|
"fs-extra": "^11.2.0",
|
|
79
|
-
"gaxios": "^6.
|
|
79
|
+
"gaxios": "^6.6.0",
|
|
80
80
|
"graceful-fs": "^4.2.11",
|
|
81
81
|
"html-escaper": "^3.0.3",
|
|
82
82
|
"html-to-text": "^9.0.5",
|
|
83
83
|
"import-meta-resolve": "^4.1.0",
|
|
84
|
-
"jieba-wasm": "^0.0
|
|
85
|
-
"jsdom": "^24.
|
|
84
|
+
"jieba-wasm": "^1.0.0",
|
|
85
|
+
"jsdom": "^24.1.0",
|
|
86
86
|
"json5": "^2.2.3",
|
|
87
87
|
"kuromoji": "^0.1.2",
|
|
88
88
|
"microsoft-cognitiveservices-speech-sdk": "^1.36.0",
|
|
89
89
|
"moving-median": "^1.0.0",
|
|
90
90
|
"msgpack-lite": "^0.1.26",
|
|
91
|
-
"onnxruntime-node": "^1.
|
|
92
|
-
"openai": "^4.
|
|
91
|
+
"onnxruntime-node": "^1.18.0",
|
|
92
|
+
"openai": "^4.47.1",
|
|
93
93
|
"sam-js": "^0.2.1",
|
|
94
94
|
"strip-ansi": "^7.1.0",
|
|
95
95
|
"tar": "^7.1.0",
|
|
96
|
-
"tiktoken": "^1.0.
|
|
96
|
+
"tiktoken": "^1.0.15",
|
|
97
97
|
"tinyld": "^1.3.4",
|
|
98
98
|
"ws": "^8.17.0",
|
|
99
99
|
"wtf_wikipedia": "^10.3.1"
|
|
@@ -120,11 +120,11 @@
|
|
|
120
120
|
"@types/graceful-fs": "^4.1.9",
|
|
121
121
|
"@types/jsdom": "^21.1.6",
|
|
122
122
|
"@types/msgpack-lite": "^0.1.11",
|
|
123
|
-
"@types/node": "^20.12.
|
|
123
|
+
"@types/node": "^20.12.12",
|
|
124
124
|
"@types/recursive-readdir": "^2.2.4",
|
|
125
125
|
"@types/tar": "^6.1.13",
|
|
126
126
|
"@types/ws": "^8.5.10",
|
|
127
|
-
"ts-json-schema-generator": "^2.
|
|
127
|
+
"ts-json-schema-generator": "^2.2.0",
|
|
128
128
|
"typescript": "^5.4.5"
|
|
129
129
|
}
|
|
130
130
|
}
|
|
@@ -1,21 +1,27 @@
|
|
|
1
|
-
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclidianDistance, magnitude } from '../math/VectorMath.js'
|
|
1
|
+
import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclidianDistance, euclidianDistance13Dim, magnitude } from '../math/VectorMath.js'
|
|
2
2
|
import { logToStderr } from '../utilities/Utilities.js'
|
|
3
3
|
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
4
4
|
|
|
5
5
|
const log = logToStderr
|
|
6
6
|
|
|
7
|
-
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number,
|
|
8
|
-
if (
|
|
7
|
+
export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidian' | 'cosine' = 'euclidian', centerIndexes?: number[]) {
|
|
8
|
+
if (distanceFunctionKind == 'euclidian') {
|
|
9
|
+
let distanceFunction = euclidianDistance
|
|
10
|
+
|
|
11
|
+
if (mfccFrames1.length > 0 && mfccFrames1[0].length === 13) {
|
|
12
|
+
distanceFunction = euclidianDistance13Dim
|
|
13
|
+
}
|
|
14
|
+
|
|
9
15
|
const { path } = alignDTWWindowed(
|
|
10
16
|
mfccFrames1,
|
|
11
17
|
mfccFrames2,
|
|
12
|
-
|
|
18
|
+
distanceFunction,
|
|
13
19
|
windowLength,
|
|
14
20
|
centerIndexes
|
|
15
21
|
)
|
|
16
22
|
|
|
17
23
|
return path
|
|
18
|
-
} else if (
|
|
24
|
+
} else if (distanceFunctionKind == 'cosine') {
|
|
19
25
|
const indexes1 = createVectorForIntegerRange(0, mfccFrames1.length)
|
|
20
26
|
const indexes2 = createVectorForIntegerRange(0, mfccFrames2.length)
|
|
21
27
|
|
|
@@ -93,7 +93,7 @@ function computeAccumulatedCostMatrixTransposed<T, U>(sequence1: T[], sequence2:
|
|
|
93
93
|
// and left column's window offset
|
|
94
94
|
const windowOffsetDelta = windowStartOffset - windowStartOffsets[columnIndex - 1]
|
|
95
95
|
|
|
96
|
-
//
|
|
96
|
+
// Compute the accumulated cost for all rows in the window
|
|
97
97
|
for (let rowIndex = 0; rowIndex < rowCount; rowIndex++) {
|
|
98
98
|
// Compute the cost for current cell
|
|
99
99
|
const cost = costFunction(targetSequence1Value, sequence2[windowStartOffset + rowIndex])
|
|
@@ -167,7 +167,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
|
|
|
167
167
|
let upCost = Infinity
|
|
168
168
|
|
|
169
169
|
if (upRowIndex >= 0) {
|
|
170
|
-
upCost = accumulatedCostMatrixTransposed[columnIndex][upRowIndex]
|
|
170
|
+
upCost = accumulatedCostMatrixTransposed[columnIndex][upRowIndex]
|
|
171
171
|
}
|
|
172
172
|
|
|
173
173
|
// Retrieve the cost for the 'left' (deletion) neighbor
|
|
@@ -176,7 +176,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
|
|
|
176
176
|
let leftCost = Infinity
|
|
177
177
|
|
|
178
178
|
if (leftColumnIndex >= 0 && leftRowIndex < rowCount) {
|
|
179
|
-
leftCost = accumulatedCostMatrixTransposed[leftColumnIndex][leftRowIndex]
|
|
179
|
+
leftCost = accumulatedCostMatrixTransposed[leftColumnIndex][leftRowIndex]
|
|
180
180
|
}
|
|
181
181
|
|
|
182
182
|
// Retrieve the cost for the 'up and left' (match) neighbor
|
|
@@ -185,7 +185,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
|
|
|
185
185
|
let upAndLeftCost = Infinity
|
|
186
186
|
|
|
187
187
|
if (upAndLeftColumnIndex >= 0 && upAndLeftRowIndex >= 0 && upAndLeftRowIndex < rowCount) {
|
|
188
|
-
upAndLeftCost = accumulatedCostMatrixTransposed[upAndLeftColumnIndex][upAndLeftRowIndex]
|
|
188
|
+
upAndLeftCost = accumulatedCostMatrixTransposed[upAndLeftColumnIndex][upAndLeftRowIndex]
|
|
189
189
|
}
|
|
190
190
|
|
|
191
191
|
// If all neighbors have a cost of infinity, it means
|
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
import { type PreTrainedModel, type PreTrainedTokenizer } from '@echogarden/transformers-nodejs-lite'
|
|
2
|
+
import { Logger } from '../utilities/Logger.js'
|
|
3
|
+
import { loadPackage } from '../utilities/PackageManager.js'
|
|
4
|
+
import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
|
|
5
|
+
import { cosineDistance } from '../math/VectorMath.js'
|
|
6
|
+
import { isPunctuation, isWord, splitToSentences, splitToWords } from '../nlp/Segmentation.js'
|
|
7
|
+
import { Timeline, extractEntries } from '../utilities/Timeline.js'
|
|
8
|
+
|
|
9
|
+
export async function alignTimelineToTextSemantically(timeline: Timeline, text: string, textLangCode: string) {
|
|
10
|
+
const logger = new Logger()
|
|
11
|
+
|
|
12
|
+
logger.start(`Prepare text for semantic alignment`)
|
|
13
|
+
|
|
14
|
+
const timelineSentenceEntries = extractEntries(timeline, entry => entry.type === 'sentence')
|
|
15
|
+
|
|
16
|
+
const timelineWordEntryGroups: Timeline[] = []
|
|
17
|
+
const timelineWordGroups: string[][] = []
|
|
18
|
+
|
|
19
|
+
for (const sentenceEntry of timelineSentenceEntries) {
|
|
20
|
+
const wordEntryGroup = sentenceEntry.timeline!
|
|
21
|
+
.filter(wordEntry => isWord(wordEntry.text))
|
|
22
|
+
|
|
23
|
+
timelineWordEntryGroups.push(wordEntryGroup)
|
|
24
|
+
timelineWordGroups.push(wordEntryGroup.map(wordEntry => wordEntry.text))
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const timelineWordEntriesFiltered = timelineWordEntryGroups.flat()
|
|
28
|
+
|
|
29
|
+
const textSentences = splitToSentences(text, textLangCode)
|
|
30
|
+
|
|
31
|
+
const textWordGroups: string[][] = []
|
|
32
|
+
|
|
33
|
+
for (const sentenceText of textSentences) {
|
|
34
|
+
let wordGroup = await splitToWords(sentenceText, textLangCode)
|
|
35
|
+
wordGroup = wordGroup.filter(word => isWord(word))
|
|
36
|
+
|
|
37
|
+
textWordGroups.push(wordGroup)
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const textWords = textWordGroups.flat()
|
|
41
|
+
|
|
42
|
+
logger.end()
|
|
43
|
+
|
|
44
|
+
const wordMappingEntries = await alignWordsToWordsSemantically(timelineWordGroups, textWordGroups)
|
|
45
|
+
|
|
46
|
+
logger.start(`Build timeline for translation`)
|
|
47
|
+
|
|
48
|
+
const mappingGroups = new Map<number, number[]>()
|
|
49
|
+
|
|
50
|
+
for (const wordMappingEntry of wordMappingEntries) {
|
|
51
|
+
const wordIndex1 = wordMappingEntry.wordIndex1
|
|
52
|
+
const wordIndex2 = wordMappingEntry.wordIndex2
|
|
53
|
+
|
|
54
|
+
let group = mappingGroups.get(wordIndex1)
|
|
55
|
+
|
|
56
|
+
if (!group) {
|
|
57
|
+
group = []
|
|
58
|
+
mappingGroups.set(wordIndex1, group)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
if (!group.includes(wordIndex2)) {
|
|
62
|
+
group.push(wordIndex2)
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
type TimeSlice = { startTime: number, endTime: number }
|
|
67
|
+
|
|
68
|
+
const timeSlicesLookup = new Map<number, TimeSlice[]>()
|
|
69
|
+
|
|
70
|
+
for (const [wordIndex1, mappedWordIndexes] of mappingGroups) {
|
|
71
|
+
if (mappedWordIndexes.length === 0) {
|
|
72
|
+
continue
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const startTime = timelineWordEntriesFiltered[wordIndex1].startTime
|
|
76
|
+
const endTime = timelineWordEntriesFiltered[wordIndex1].endTime
|
|
77
|
+
|
|
78
|
+
const splitCount = mappedWordIndexes.length
|
|
79
|
+
|
|
80
|
+
const sliceDuration = (endTime - startTime) / splitCount
|
|
81
|
+
|
|
82
|
+
let timeOffset = 0
|
|
83
|
+
|
|
84
|
+
for (let i = 0; i < splitCount; i++) {
|
|
85
|
+
const timeSlice: TimeSlice = {
|
|
86
|
+
startTime: startTime + timeOffset,
|
|
87
|
+
endTime: startTime + timeOffset + sliceDuration
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const wordIndex2 = mappedWordIndexes[i]
|
|
91
|
+
|
|
92
|
+
let timeSlicesForTargetWord = timeSlicesLookup.get(wordIndex2)
|
|
93
|
+
|
|
94
|
+
if (!timeSlicesForTargetWord) {
|
|
95
|
+
timeSlicesForTargetWord = []
|
|
96
|
+
timeSlicesLookup.set(wordIndex2, timeSlicesForTargetWord)
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
timeSlicesForTargetWord.push(timeSlice)
|
|
100
|
+
|
|
101
|
+
timeOffset += sliceDuration
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
const resultTimeline: Timeline = []
|
|
106
|
+
|
|
107
|
+
for (const [key, value] of timeSlicesLookup) {
|
|
108
|
+
resultTimeline.push({
|
|
109
|
+
type: 'word',
|
|
110
|
+
text: textWords[key],
|
|
111
|
+
startTime: value[0].startTime,
|
|
112
|
+
endTime: value[value.length - 1].endTime
|
|
113
|
+
})
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
logger.end()
|
|
117
|
+
|
|
118
|
+
return resultTimeline
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export async function alignWordsToWordsSemantically(wordsGroups1: string[][], wordsGroups2: string[][], windowTokenCount = 1000 * 1000) {
|
|
122
|
+
const logger = new Logger()
|
|
123
|
+
|
|
124
|
+
// Load embedding model
|
|
125
|
+
const modelPath = await loadPackage(`xenova-multilingual-e5-small-fp16`)
|
|
126
|
+
|
|
127
|
+
const embeddingModel = new E5TextEmbedding(modelPath)
|
|
128
|
+
|
|
129
|
+
logger.start(`Initialize E5 embedding model`)
|
|
130
|
+
await embeddingModel.initializeIfNeeded()
|
|
131
|
+
|
|
132
|
+
async function extractEmbeddingsFromWordGroups(wordGroups: string[][]) {
|
|
133
|
+
const logger = new Logger()
|
|
134
|
+
|
|
135
|
+
const maxTokensPerFragment = 512
|
|
136
|
+
const { Tensor } = await import('@echogarden/transformers-nodejs-lite')
|
|
137
|
+
|
|
138
|
+
const words: string[] = []
|
|
139
|
+
|
|
140
|
+
const embeddings: TokenEmbeddingData[] = []
|
|
141
|
+
const tokenToWordIndexMapping: number[] = []
|
|
142
|
+
|
|
143
|
+
for (const wordGroup of wordGroups) {
|
|
144
|
+
const { joinedText: joinedTextForGroup, offsets: offsetsForGroup } = joinAndGetOffsets(wordGroup)
|
|
145
|
+
|
|
146
|
+
logger.start(`Tokenize text`)
|
|
147
|
+
const inputsForGroup = await embeddingModel.tokenizeToModelInputs(joinedTextForGroup)
|
|
148
|
+
|
|
149
|
+
logger.start(`Infer embeddings for text`)
|
|
150
|
+
|
|
151
|
+
const allTokenIds = inputsForGroup['input_ids'].data
|
|
152
|
+
const allAttentionMask = inputsForGroup['attention_mask'].data
|
|
153
|
+
|
|
154
|
+
let embeddingsForGroup: TokenEmbeddingData[] = []
|
|
155
|
+
|
|
156
|
+
for (let tokenStart = 0; tokenStart < allTokenIds.length; tokenStart += maxTokensPerFragment) {
|
|
157
|
+
const tokenEnd = Math.min(tokenStart + maxTokensPerFragment, allTokenIds.length)
|
|
158
|
+
const fragmentTokenCount = tokenEnd - tokenStart
|
|
159
|
+
|
|
160
|
+
const fragmentInputIdsTensor = new Tensor('int64', allTokenIds.slice(tokenStart, tokenEnd), [1, fragmentTokenCount])
|
|
161
|
+
const fragmentAttentionMaskTensor = new Tensor('int64', allAttentionMask.slice(tokenStart, tokenEnd), [1, fragmentTokenCount])
|
|
162
|
+
|
|
163
|
+
const inputsForFragment = { input_ids: fragmentInputIdsTensor, attention_mask: fragmentAttentionMaskTensor }
|
|
164
|
+
|
|
165
|
+
const embeddingsForFragment = await embeddingModel.inferTokenEmbeddings(inputsForFragment)
|
|
166
|
+
|
|
167
|
+
embeddingsForGroup.push(...embeddingsForFragment)
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
logger.start(`Compute token to word mapping for text`)
|
|
171
|
+
const filteredEmbeddingsForGroup = embeddingsForGroup.filter((embedding) => embedding.text !== '▁' && embedding.text !== '<s>' && embedding.text !== '</s>')
|
|
172
|
+
const tokenToWordIndexMappingForGroup = mapTokenEmbeddingsToWordIndexes(filteredEmbeddingsForGroup, joinedTextForGroup, offsetsForGroup)
|
|
173
|
+
const tokenToWordIndexMappingForGroupWithOffset = tokenToWordIndexMappingForGroup.map(value => words.length + value)
|
|
174
|
+
|
|
175
|
+
embeddings.push(...filteredEmbeddingsForGroup)
|
|
176
|
+
tokenToWordIndexMapping.push(...tokenToWordIndexMappingForGroupWithOffset)
|
|
177
|
+
|
|
178
|
+
words.push(...wordGroup)
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
return { words, embeddings, tokenToWordIndexMapping }
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
logger.start(`Extract embeddings from source 1`)
|
|
185
|
+
const {
|
|
186
|
+
words: words1,
|
|
187
|
+
embeddings: embeddings1,
|
|
188
|
+
tokenToWordIndexMapping: tokenToWordIndexMapping1
|
|
189
|
+
} = await extractEmbeddingsFromWordGroups(wordsGroups1)
|
|
190
|
+
|
|
191
|
+
logger.start(`Extract embeddings from source 2`)
|
|
192
|
+
const {
|
|
193
|
+
words: words2,
|
|
194
|
+
embeddings: embeddings2,
|
|
195
|
+
tokenToWordIndexMapping: tokenToWordIndexMapping2
|
|
196
|
+
} = await extractEmbeddingsFromWordGroups(wordsGroups2)
|
|
197
|
+
|
|
198
|
+
// Align
|
|
199
|
+
function costFunction(a: TokenEmbeddingData, b: TokenEmbeddingData) {
|
|
200
|
+
const aIsPunctuation = isPunctuation(a.text)
|
|
201
|
+
const bIsPunctuation = isPunctuation(b.text)
|
|
202
|
+
|
|
203
|
+
if (aIsPunctuation === bIsPunctuation) {
|
|
204
|
+
return cosineDistance(a.embeddingVector, b.embeddingVector)
|
|
205
|
+
} else {
|
|
206
|
+
return 1.0
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
logger.start(`Align token embedding vectors using DTW`)
|
|
211
|
+
|
|
212
|
+
const { path } = alignDTWWindowed(embeddings1, embeddings2, costFunction, windowTokenCount)
|
|
213
|
+
|
|
214
|
+
// Use alignment path to words to words
|
|
215
|
+
logger.start(`Map tokens to words`)
|
|
216
|
+
|
|
217
|
+
const wordMapping: WordMapping[] = []
|
|
218
|
+
|
|
219
|
+
for (let i = 0; i < path.length; i++) {
|
|
220
|
+
const pathEntry = path[i]
|
|
221
|
+
|
|
222
|
+
const sourceTokenIndex = pathEntry.source
|
|
223
|
+
const destTokenIndex = pathEntry.dest
|
|
224
|
+
|
|
225
|
+
const mappedWordIndex1 = tokenToWordIndexMapping1[sourceTokenIndex]
|
|
226
|
+
const mappedWordIndex2 = tokenToWordIndexMapping2[destTokenIndex]
|
|
227
|
+
|
|
228
|
+
wordMapping.push({
|
|
229
|
+
wordIndex1: mappedWordIndex1,
|
|
230
|
+
word1: words1[mappedWordIndex1],
|
|
231
|
+
wordIndex2: mappedWordIndex2,
|
|
232
|
+
word2: words2[mappedWordIndex2],
|
|
233
|
+
})
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
logger.end()
|
|
237
|
+
|
|
238
|
+
return wordMapping
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
function mapTokenEmbeddingsToWordIndexes(embeddings: TokenEmbeddingData[], text: string, textWordOffsets: number[]) {
|
|
242
|
+
const tokenToWordIndex: number[] = []
|
|
243
|
+
|
|
244
|
+
let currentTextOffset = 0
|
|
245
|
+
|
|
246
|
+
for (let i = 0; i < embeddings.length; i++) {
|
|
247
|
+
const embedding = embeddings[i]
|
|
248
|
+
let tokenText = embedding.text
|
|
249
|
+
|
|
250
|
+
if (tokenText === '<s>' || tokenText === '</s>') {
|
|
251
|
+
tokenToWordIndex.push(-1)
|
|
252
|
+
|
|
253
|
+
continue
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (tokenText.startsWith('▁')) {
|
|
257
|
+
tokenText = tokenText.substring(1)
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
const matchPosition = text.indexOf(tokenText, currentTextOffset)
|
|
261
|
+
|
|
262
|
+
if (matchPosition === -1) {
|
|
263
|
+
throw new Error(`Token '${tokenText}' not found in text`)
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
currentTextOffset = matchPosition + tokenText.length
|
|
267
|
+
|
|
268
|
+
let tokenMatchingWordIndex = textWordOffsets.findIndex((index) => index > matchPosition)
|
|
269
|
+
|
|
270
|
+
if (tokenMatchingWordIndex === -1) {
|
|
271
|
+
throw new Error(`Token '${tokenText}' not found in text`)
|
|
272
|
+
} else {
|
|
273
|
+
tokenMatchingWordIndex = Math.max(tokenMatchingWordIndex - 1, 0)
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
tokenToWordIndex.push(tokenMatchingWordIndex)
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
return tokenToWordIndex
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
function joinAndGetOffsets(words: string[]) {
|
|
283
|
+
let joinedText = ''
|
|
284
|
+
const offsets: number[] = []
|
|
285
|
+
|
|
286
|
+
let offset = 0
|
|
287
|
+
|
|
288
|
+
for (const word of words) {
|
|
289
|
+
const extendedWord = `${word} `
|
|
290
|
+
joinedText += extendedWord
|
|
291
|
+
|
|
292
|
+
offsets.push(offset)
|
|
293
|
+
|
|
294
|
+
offset += extendedWord.length
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
offsets.push(joinedText.length)
|
|
298
|
+
|
|
299
|
+
return { joinedText, offsets }
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
export class E5TextEmbedding {
|
|
303
|
+
tokenizer?: PreTrainedTokenizer
|
|
304
|
+
model?: PreTrainedModel
|
|
305
|
+
|
|
306
|
+
constructor(public readonly modelPath: string) {
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
async tokenizeToModelInputs(text: string) {
|
|
310
|
+
await this.initializeIfNeeded()
|
|
311
|
+
|
|
312
|
+
const inputs = await this.tokenizer!(text)
|
|
313
|
+
|
|
314
|
+
return inputs
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
async inferTokenEmbeddings(inputs: any) {
|
|
318
|
+
await this.initializeIfNeeded()
|
|
319
|
+
|
|
320
|
+
const tokensText = this.tokenizer!.model.convert_ids_to_tokens(Array.from(inputs.input_ids.data))
|
|
321
|
+
|
|
322
|
+
const result = await this.model!(inputs)
|
|
323
|
+
|
|
324
|
+
const lastHiddenState = result.last_hidden_state
|
|
325
|
+
|
|
326
|
+
const tokenCount = lastHiddenState.dims[1]
|
|
327
|
+
const embeddingSize = lastHiddenState.dims[2]
|
|
328
|
+
|
|
329
|
+
const tokenEmbeddings: TokenEmbeddingData[] = []
|
|
330
|
+
|
|
331
|
+
for (let i = 0; i < tokenCount; i++) {
|
|
332
|
+
const tokenEmbeddingVector = lastHiddenState.data.slice(i * embeddingSize, (i + 1) * embeddingSize)
|
|
333
|
+
|
|
334
|
+
const tokenId = Number(inputs.input_ids.data[i])
|
|
335
|
+
const tokenText = tokensText[i]
|
|
336
|
+
|
|
337
|
+
tokenEmbeddings.push({
|
|
338
|
+
id: tokenId,
|
|
339
|
+
text: tokenText,
|
|
340
|
+
embeddingVector: tokenEmbeddingVector
|
|
341
|
+
})
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
return tokenEmbeddings
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
async initializeIfNeeded() {
|
|
348
|
+
if (this.tokenizer && this.model) {
|
|
349
|
+
return
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
const { AutoTokenizer, AutoModel } = await import('@echogarden/transformers-nodejs-lite')
|
|
353
|
+
|
|
354
|
+
this.tokenizer = await AutoTokenizer.from_pretrained(this.modelPath)
|
|
355
|
+
this.model = await AutoModel.from_pretrained(this.modelPath)
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
export interface TokenEmbeddingData {
|
|
360
|
+
id: number
|
|
361
|
+
text: string
|
|
362
|
+
embeddingVector: Float32Array
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
export interface WordMapping {
|
|
366
|
+
wordIndex1: number
|
|
367
|
+
word1: string
|
|
368
|
+
|
|
369
|
+
wordIndex2: number
|
|
370
|
+
word2: string
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
export const e5SupportedLanguages: string[] = [
|
|
374
|
+
'af', // Afrikaans
|
|
375
|
+
'am', // Amharic
|
|
376
|
+
'ar', // Arabic
|
|
377
|
+
'as', // Assamese
|
|
378
|
+
'az', // Azerbaijani
|
|
379
|
+
'be', // Belarusian
|
|
380
|
+
'bg', // Bulgarian
|
|
381
|
+
'bn', // Bengali
|
|
382
|
+
'br', // Breton
|
|
383
|
+
'bs', // Bosnian
|
|
384
|
+
'ca', // Catalan
|
|
385
|
+
'cs', // Czech
|
|
386
|
+
'cy', // Welsh
|
|
387
|
+
'da', // Danish
|
|
388
|
+
'de', // German
|
|
389
|
+
'el', // Greek
|
|
390
|
+
'en', // English
|
|
391
|
+
'eo', // Esperanto
|
|
392
|
+
'es', // Spanish
|
|
393
|
+
'et', // Estonian
|
|
394
|
+
'eu', // Basque
|
|
395
|
+
'fa', // Persian
|
|
396
|
+
'fi', // Finnish
|
|
397
|
+
'fr', // French
|
|
398
|
+
'fy', // Western Frisian
|
|
399
|
+
'ga', // Irish
|
|
400
|
+
'gd', // Scottish Gaelic
|
|
401
|
+
'gl', // Galician
|
|
402
|
+
'gu', // Gujarati
|
|
403
|
+
'ha', // Hausa
|
|
404
|
+
'he', // Hebrew
|
|
405
|
+
'hi', // Hindi
|
|
406
|
+
'hr', // Croatian
|
|
407
|
+
'hu', // Hungarian
|
|
408
|
+
'hy', // Armenian
|
|
409
|
+
'id', // Indonesian
|
|
410
|
+
'is', // Icelandic
|
|
411
|
+
'it', // Italian
|
|
412
|
+
'ja', // Japanese
|
|
413
|
+
'jv', // Javanese
|
|
414
|
+
'ka', // Georgian
|
|
415
|
+
'kk', // Kazakh
|
|
416
|
+
'km', // Khmer
|
|
417
|
+
'kn', // Kannada
|
|
418
|
+
'ko', // Korean
|
|
419
|
+
'ku', // Kurdish
|
|
420
|
+
'ky', // Kyrgyz
|
|
421
|
+
'la', // Latin
|
|
422
|
+
'lo', // Lao
|
|
423
|
+
'lt', // Lithuanian
|
|
424
|
+
'lv', // Latvian
|
|
425
|
+
'mg', // Malagasy
|
|
426
|
+
'mk', // Macedonian
|
|
427
|
+
'ml', // Malayalam
|
|
428
|
+
'mn', // Mongolian
|
|
429
|
+
'mr', // Marathi
|
|
430
|
+
'ms', // Malay
|
|
431
|
+
'my', // Burmese
|
|
432
|
+
'ne', // Nepali
|
|
433
|
+
'nl', // Dutch
|
|
434
|
+
'no', // Norwegian
|
|
435
|
+
'om', // Oromo
|
|
436
|
+
'or', // Oriya
|
|
437
|
+
'pa', // Panjabi
|
|
438
|
+
'pl', // Polish
|
|
439
|
+
'ps', // Pashto
|
|
440
|
+
'pt', // Portuguese
|
|
441
|
+
'ro', // Romanian
|
|
442
|
+
'ru', // Russian
|
|
443
|
+
'sa', // Sanskrit
|
|
444
|
+
'sd', // Sindhi
|
|
445
|
+
'si', // Sinhala
|
|
446
|
+
'sk', // Slovak
|
|
447
|
+
'sl', // Slovenian
|
|
448
|
+
'so', // Somali
|
|
449
|
+
'sq', // Albanian
|
|
450
|
+
'sr', // Serbian
|
|
451
|
+
'su', // Sundanese
|
|
452
|
+
'sv', // Swedish
|
|
453
|
+
'sw', // Swahili
|
|
454
|
+
'ta', // Tamil
|
|
455
|
+
'te', // Telugu
|
|
456
|
+
'th', // Thai
|
|
457
|
+
'tl', // Tagalog
|
|
458
|
+
'tr', // Turkish
|
|
459
|
+
'ug', // Uyghur
|
|
460
|
+
'uk', // Ukrainian
|
|
461
|
+
'ur', // Urdu
|
|
462
|
+
'uz', // Uzbek
|
|
463
|
+
'vi', // Vietnamese
|
|
464
|
+
'xh', // Xhosa
|
|
465
|
+
'yi', // Yiddish
|
|
466
|
+
'zh', // Chinese
|
|
467
|
+
]
|