echogarden 1.4.3 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/README.md +1 -1
  2. package/data/schemas/options.json +267 -19
  3. package/data/tables/lcid-table.json +9 -0
  4. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
  5. package/dist/alignment/DTWMfccSequenceAlignment.js +9 -5
  6. package/dist/alignment/DTWMfccSequenceAlignment.js.map +1 -1
  7. package/dist/alignment/DTWSequenceAlignmentWindowed.js +4 -4
  8. package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
  9. package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +10 -1
  10. package/dist/alignment/SemanticTextAlignment.js +336 -0
  11. package/dist/alignment/SemanticTextAlignment.js.map +1 -0
  12. package/dist/alignment/SpeechAlignment.d.ts +2 -1
  13. package/dist/alignment/SpeechAlignment.js +106 -38
  14. package/dist/alignment/SpeechAlignment.js.map +1 -1
  15. package/dist/api/API.d.ts +7 -2
  16. package/dist/api/API.js +7 -2
  17. package/dist/api/API.js.map +1 -1
  18. package/dist/api/APIOptions.d.ts +4 -1
  19. package/dist/api/Alignment.d.ts +1 -1
  20. package/dist/api/Alignment.js +14 -6
  21. package/dist/api/Alignment.js.map +1 -1
  22. package/dist/api/LanguageDetectionCommon.d.ts +6 -0
  23. package/dist/api/LanguageDetectionCommon.js +2 -0
  24. package/dist/api/LanguageDetectionCommon.js.map +1 -0
  25. package/dist/api/Recognition.js.map +1 -1
  26. package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
  27. package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
  28. package/dist/api/SpeechLanguageDetection.js.map +1 -0
  29. package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
  30. package/dist/api/SpeechTranslation.js.map +1 -0
  31. package/dist/api/Synthesis.js +4 -4
  32. package/dist/api/TextLanguageDetection.d.ts +21 -0
  33. package/dist/api/TextLanguageDetection.js +67 -0
  34. package/dist/api/TextLanguageDetection.js.map +1 -0
  35. package/dist/api/TextTranslation.d.ts +5 -0
  36. package/dist/api/TextTranslation.js +4 -0
  37. package/dist/api/TextTranslation.js.map +1 -0
  38. package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
  39. package/dist/api/TimelineTranslationAlignment.js +92 -0
  40. package/dist/api/TimelineTranslationAlignment.js.map +1 -0
  41. package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
  42. package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
  43. package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
  44. package/dist/api/TranslationAlignment.d.ts +4 -3
  45. package/dist/api/TranslationAlignment.js +9 -8
  46. package/dist/api/TranslationAlignment.js.map +1 -1
  47. package/dist/api/VoiceActivityDetection.js +16 -1
  48. package/dist/api/VoiceActivityDetection.js.map +1 -1
  49. package/dist/cli/CLI.d.ts +27 -7
  50. package/dist/cli/CLI.js +205 -34
  51. package/dist/cli/CLI.js.map +1 -1
  52. package/dist/codecs/FFMpegTranscoder.js +7 -0
  53. package/dist/codecs/FFMpegTranscoder.js.map +1 -1
  54. package/dist/dsp/FFT.d.ts +1 -1
  55. package/dist/dsp/FFT.js +6 -0
  56. package/dist/dsp/FFT.js.map +1 -1
  57. package/dist/dsp/KWeightingFilter.js +1 -1
  58. package/dist/dsp/KWeightingFilter.js.map +1 -1
  59. package/dist/dsp/MelSpectogram.d.ts +3 -2
  60. package/dist/dsp/MelSpectogram.js +14 -8
  61. package/dist/dsp/MelSpectogram.js.map +1 -1
  62. package/dist/math/VectorMath.d.ts +22 -20
  63. package/dist/math/VectorMath.js +57 -30
  64. package/dist/math/VectorMath.js.map +1 -1
  65. package/dist/recognition/WhisperCppSTT.d.ts +1 -1
  66. package/dist/recognition/WhisperCppSTT.js +2 -2
  67. package/dist/recognition/WhisperCppSTT.js.map +1 -1
  68. package/dist/recognition/WhisperSTT.js +9 -6
  69. package/dist/recognition/WhisperSTT.js.map +1 -1
  70. package/dist/server/Client.d.ts +3 -2
  71. package/dist/server/Client.js.map +1 -1
  72. package/dist/server/Worker.d.ts +3 -2
  73. package/dist/server/Worker.js +3 -2
  74. package/dist/server/Worker.js.map +1 -1
  75. package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
  76. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
  77. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +13 -0
  78. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
  79. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
  80. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -1
  81. package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
  82. package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
  83. package/dist/synthesis/EspeakTTS.js +4 -5
  84. package/dist/synthesis/EspeakTTS.js.map +1 -1
  85. package/dist/tests/Test.js +0 -8
  86. package/dist/tests/Test.js.map +1 -1
  87. package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
  88. package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
  89. package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
  90. package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
  91. package/dist/text-translation/NLLBTextTranslation.js +1 -1
  92. package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
  93. package/dist/utilities/Locale.d.ts +1 -1
  94. package/dist/utilities/Locale.js +1 -1
  95. package/dist/utilities/OnnxUtilities.d.ts +1 -1
  96. package/dist/utilities/PackageManager.js +8 -2
  97. package/dist/utilities/PackageManager.js.map +1 -1
  98. package/dist/utilities/Timeline.d.ts +1 -0
  99. package/dist/utilities/Timeline.js +12 -0
  100. package/dist/utilities/Timeline.js.map +1 -1
  101. package/docs/API.md +81 -3
  102. package/docs/CLI.md +51 -1
  103. package/docs/Engines.md +24 -0
  104. package/docs/Options.md +33 -1
  105. package/docs/Tasklist.md +4 -1
  106. package/package.json +11 -11
  107. package/src/alignment/DTWMfccSequenceAlignment.ts +11 -5
  108. package/src/alignment/DTWSequenceAlignmentWindowed.ts +4 -4
  109. package/src/alignment/SemanticTextAlignment.ts +467 -0
  110. package/src/alignment/SpeechAlignment.ts +180 -53
  111. package/src/api/API.ts +18 -2
  112. package/src/api/APIOptions.ts +14 -1
  113. package/src/api/Alignment.ts +32 -10
  114. package/src/api/LanguageDetectionCommon.ts +7 -0
  115. package/src/api/Recognition.ts +2 -0
  116. package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
  117. package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
  118. package/src/api/Synthesis.ts +4 -4
  119. package/src/api/TextLanguageDetection.ts +116 -0
  120. package/src/api/TextTranslation.ts +9 -0
  121. package/src/api/TimelineTranslationAlignment.ts +162 -0
  122. package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
  123. package/src/api/TranslationAlignment.ts +12 -10
  124. package/src/api/VoiceActivityDetection.ts +24 -3
  125. package/src/cli/CLI.ts +276 -34
  126. package/src/codecs/FFMpegTranscoder.ts +6 -0
  127. package/src/dsp/FFT.ts +8 -2
  128. package/src/dsp/KWeightingFilter.ts +1 -1
  129. package/src/dsp/MelSpectogram.ts +17 -8
  130. package/src/math/VectorMath.ts +87 -52
  131. package/src/recognition/WhisperCppSTT.ts +2 -2
  132. package/src/recognition/WhisperSTT.ts +10 -6
  133. package/src/server/Client.ts +3 -2
  134. package/src/server/Worker.ts +3 -2
  135. package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
  136. package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
  137. package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
  138. package/src/synthesis/EspeakTTS.ts +6 -8
  139. package/src/tests/Test.ts +1 -12
  140. package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
  141. package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
  142. package/src/text-translation/NLLBTextTranslation.ts +1 -1
  143. package/src/utilities/Locale.ts +1 -1
  144. package/src/utilities/OnnxUtilities.ts +1 -1
  145. package/src/utilities/PackageManager.ts +9 -2
  146. package/src/utilities/Timeline.ts +14 -0
  147. package/dist/alignment/TextAlignment.js +0 -63
  148. package/dist/alignment/TextAlignment.js.map +0 -1
  149. package/dist/api/LanguageDetection.js.map +0 -1
  150. package/dist/api/Translation.js.map +0 -1
  151. package/src/alignment/TextAlignment.ts +0 -96
  152. /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
package/docs/Options.md CHANGED
@@ -271,7 +271,8 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
271
271
 
272
272
  **General**:
273
273
  * `engine`: alignment algorithm to use, can only be `whisper`. Defaults to `whisper`
274
- * `language`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
274
+ * `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
275
+ * `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
275
276
  * `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
276
277
  * `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
277
278
  * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
@@ -284,6 +285,37 @@ Applies to CLI operation: `align-translation`, API method: `alignTranslation`
284
285
  * `whisper.encoderProvider`: encoder ONNX execution provider. See details in recognition section above
285
286
  * `whisper.decoderProvider`: decoder ONNX execution provider. See details in recognition section above
286
287
 
288
+ ## Speech-to-transcript-and-translation alignment
289
+
290
+ Applies to CLI operation: `align-transcript-and-translation`, API method: `alignTranscriptAndTranslation`
291
+
292
+ **General**:
293
+ * `engine`: can only be `two-stage`. Defaults to `two-stage`
294
+ * `sourceLanguage`: language code for the source audio ([ISO 639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)), like `en`, `fr`, `zh`, etc. Auto-detected from audio if not set
295
+ * `targetLanguage`: language code for the translated transcript. Can only be `en` for now. Defaults to `en`
296
+ * `crop`: crop to active parts using voice activity detection before starting. Defaults to `true`
297
+ * `isolate`: apply source separation to isolate voice before starting alignment. Defaults to `false`
298
+ * `alignment`: prefix to provide options for alignment. Options detailed in section for alignment
299
+ * `timelineAlignment`: prefix to provide options for timeline alignment. Options detailed in section for timeline alignment
300
+ * `vad`: prefix to provide options for voice activity detection when `crop` is set to `true`. Options detailed in section for voice activity detection
301
+ * `sourceSeparation`: prefix to provide options for source separation when `isolate` is set to `true`. Options detailed in section for source separation
302
+ * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
303
+
304
+ ## Timeline-to-translated-text alignment
305
+
306
+ Applies to CLI operation: `align-timeline-translation`, API method: `alignTimelineTranslation`
307
+
308
+ **General**:
309
+ * `engine`: alignment engine to use. Can only be `e5`. Defaults to `e5`
310
+ * `sourceLanguage`: language code for the source timeline. Auto-detected from timeline if not set
311
+ * `targetLanguage`: language code for the translated transcript. Auto-detected if not set
312
+ * `audio`: spoken audio to play when previewing the result in the CLI (not required or used by the alignment itself). Optional
313
+ * `languageDetection`: prefix to provide options for language detection. Options detailed in section for text language detection
314
+ * `subtitles`: prefix to provide options for subtitles. Options detailed in section for subtitles
315
+
316
+ **E5**:
317
+ * `e5.model`: E5 model to use. Defaults to `e5-small-fp16` (support for additional models will be added in the future)
318
+
287
319
  ## Language detection
288
320
 
289
321
  ### Speech language detection
package/docs/Tasklist.md CHANGED
@@ -5,7 +5,6 @@
5
5
  ### Alignment / DTW-RA
6
6
 
7
7
  * In DTW-RA, a recognized transcript including something like "Question 2.What does Juan", where "2.What" has a point in the middle, is breaking playback of the timeline
8
- * DTW-RA will not work correctly with Polish language texts, due to issues with the eSpeak engine pronouncing `|` characters, which are intended to be used as separators and ignored by all other eSpeak languages
9
8
 
10
9
  ### Synthesis
11
10
 
@@ -159,6 +158,10 @@
159
158
 
160
159
  ### Alignment
161
160
 
161
+ ### Alignment / DTW
162
+ * Accept percentages like `20%` in the `windowDuration` option
163
+ * For the `granularity` option, add more granularities like `xxx-low` and `xxxx-low` (should the naming be changed? Maybe transition to a new naming scheme?)
164
+
162
165
  ### Alignment / DTW-RA
163
166
 
164
167
  ### Alignment / Whisper
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "echogarden",
3
- "version": "1.4.3",
3
+ "version": "1.5.0",
4
4
  "description": "An easy-to-use speech toolset. Includes tools for synthesis, recognition, alignment, speech translation, language detection, source separation and more.",
5
5
  "author": "Rotem Dan",
6
6
  "license": "GPL-3.0",
@@ -55,8 +55,8 @@
55
55
  "echogarden": "./dist/cli/CLILauncher.js"
56
56
  },
57
57
  "dependencies": {
58
- "@aws-sdk/client-polly": "^3.574.0",
59
- "@aws-sdk/client-transcribe-streaming": "^3.574.0",
58
+ "@aws-sdk/client-polly": "^3.583.0",
59
+ "@aws-sdk/client-transcribe-streaming": "^3.583.0",
60
60
  "@echogarden/espeak-ng-emscripten": "^0.1.2",
61
61
  "@echogarden/fasttext-wasm": "^0.1.0",
62
62
  "@echogarden/flite-wasi": "^0.1.1",
@@ -76,24 +76,24 @@
76
76
  "command-exists": "^1.2.9",
77
77
  "compromise": "^14.13.0",
78
78
  "fs-extra": "^11.2.0",
79
- "gaxios": "^6.5.0",
79
+ "gaxios": "^6.6.0",
80
80
  "graceful-fs": "^4.2.11",
81
81
  "html-escaper": "^3.0.3",
82
82
  "html-to-text": "^9.0.5",
83
83
  "import-meta-resolve": "^4.1.0",
84
- "jieba-wasm": "^0.0.2",
85
- "jsdom": "^24.0.0",
84
+ "jieba-wasm": "^1.0.0",
85
+ "jsdom": "^24.1.0",
86
86
  "json5": "^2.2.3",
87
87
  "kuromoji": "^0.1.2",
88
88
  "microsoft-cognitiveservices-speech-sdk": "^1.36.0",
89
89
  "moving-median": "^1.0.0",
90
90
  "msgpack-lite": "^0.1.26",
91
- "onnxruntime-node": "^1.17.3",
92
- "openai": "^4.45.0",
91
+ "onnxruntime-node": "^1.18.0",
92
+ "openai": "^4.47.1",
93
93
  "sam-js": "^0.2.1",
94
94
  "strip-ansi": "^7.1.0",
95
95
  "tar": "^7.1.0",
96
- "tiktoken": "^1.0.14",
96
+ "tiktoken": "^1.0.15",
97
97
  "tinyld": "^1.3.4",
98
98
  "ws": "^8.17.0",
99
99
  "wtf_wikipedia": "^10.3.1"
@@ -120,11 +120,11 @@
120
120
  "@types/graceful-fs": "^4.1.9",
121
121
  "@types/jsdom": "^21.1.6",
122
122
  "@types/msgpack-lite": "^0.1.11",
123
- "@types/node": "^20.12.11",
123
+ "@types/node": "^20.12.12",
124
124
  "@types/recursive-readdir": "^2.2.4",
125
125
  "@types/tar": "^6.1.13",
126
126
  "@types/ws": "^8.5.10",
127
- "ts-json-schema-generator": "^2.1.2-next.1",
127
+ "ts-json-schema-generator": "^2.2.0",
128
128
  "typescript": "^5.4.5"
129
129
  }
130
130
  }
@@ -1,21 +1,27 @@
1
- import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclidianDistance, magnitude } from '../math/VectorMath.js'
1
+ import { cosineDistancePrecomputedMagnitudes, createVectorForIntegerRange, euclidianDistance, euclidianDistance13Dim, magnitude } from '../math/VectorMath.js'
2
2
  import { logToStderr } from '../utilities/Utilities.js'
3
3
  import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
4
4
 
5
5
  const log = logToStderr
6
6
 
7
- export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunction: 'euclidian' | 'cosine' = 'euclidian', centerIndexes?: number[]) {
8
- if (distanceFunction == 'euclidian') {
7
+ export async function alignMFCC_DTW(mfccFrames1: number[][], mfccFrames2: number[][], windowLength: number, distanceFunctionKind: 'euclidian' | 'cosine' = 'euclidian', centerIndexes?: number[]) {
8
+ if (distanceFunctionKind == 'euclidian') {
9
+ let distanceFunction = euclidianDistance
10
+
11
+ if (mfccFrames1.length > 0 && mfccFrames1[0].length === 13) {
12
+ distanceFunction = euclidianDistance13Dim
13
+ }
14
+
9
15
  const { path } = alignDTWWindowed(
10
16
  mfccFrames1,
11
17
  mfccFrames2,
12
- euclidianDistance,
18
+ distanceFunction,
13
19
  windowLength,
14
20
  centerIndexes
15
21
  )
16
22
 
17
23
  return path
18
- } else if (distanceFunction == 'cosine') {
24
+ } else if (distanceFunctionKind == 'cosine') {
19
25
  const indexes1 = createVectorForIntegerRange(0, mfccFrames1.length)
20
26
  const indexes2 = createVectorForIntegerRange(0, mfccFrames2.length)
21
27
 
@@ -93,7 +93,7 @@ function computeAccumulatedCostMatrixTransposed<T, U>(sequence1: T[], sequence2:
93
93
  // and left column's window offset
94
94
  const windowOffsetDelta = windowStartOffset - windowStartOffsets[columnIndex - 1]
95
95
 
96
- // Iterate over all rows in the window
96
+ // Compute the accumulated cost for all rows in the window
97
97
  for (let rowIndex = 0; rowIndex < rowCount; rowIndex++) {
98
98
  // Compute the cost for current cell
99
99
  const cost = costFunction(targetSequence1Value, sequence2[windowStartOffset + rowIndex])
@@ -167,7 +167,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
167
167
  let upCost = Infinity
168
168
 
169
169
  if (upRowIndex >= 0) {
170
- upCost = accumulatedCostMatrixTransposed[columnIndex][upRowIndex] // insertion
170
+ upCost = accumulatedCostMatrixTransposed[columnIndex][upRowIndex]
171
171
  }
172
172
 
173
173
  // Retrieve the cost for the 'left' (deletion) neighbor
@@ -176,7 +176,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
176
176
  let leftCost = Infinity
177
177
 
178
178
  if (leftColumnIndex >= 0 && leftRowIndex < rowCount) {
179
- leftCost = accumulatedCostMatrixTransposed[leftColumnIndex][leftRowIndex] // deletion
179
+ leftCost = accumulatedCostMatrixTransposed[leftColumnIndex][leftRowIndex]
180
180
  }
181
181
 
182
182
  // Retrieve the cost for the 'up and left' (match) neighbor
@@ -185,7 +185,7 @@ function computeBestPathTransposed(accumulatedCostMatrixTransposed: Float32Array
185
185
  let upAndLeftCost = Infinity
186
186
 
187
187
  if (upAndLeftColumnIndex >= 0 && upAndLeftRowIndex >= 0 && upAndLeftRowIndex < rowCount) {
188
- upAndLeftCost = accumulatedCostMatrixTransposed[upAndLeftColumnIndex][upAndLeftRowIndex] // match
188
+ upAndLeftCost = accumulatedCostMatrixTransposed[upAndLeftColumnIndex][upAndLeftRowIndex]
189
189
  }
190
190
 
191
191
  // If all neighbors have a cost of infinity, it means
@@ -0,0 +1,467 @@
1
+ import { type PreTrainedModel, type PreTrainedTokenizer } from '@echogarden/transformers-nodejs-lite'
2
+ import { Logger } from '../utilities/Logger.js'
3
+ import { loadPackage } from '../utilities/PackageManager.js'
4
+ import { alignDTWWindowed } from './DTWSequenceAlignmentWindowed.js'
5
+ import { cosineDistance } from '../math/VectorMath.js'
6
+ import { isPunctuation, isWord, splitToSentences, splitToWords } from '../nlp/Segmentation.js'
7
+ import { Timeline, extractEntries } from '../utilities/Timeline.js'
8
+
9
+ export async function alignTimelineToTextSemantically(timeline: Timeline, text: string, textLangCode: string) {
10
+ const logger = new Logger()
11
+
12
+ logger.start(`Prepare text for semantic alignment`)
13
+
14
+ const timelineSentenceEntries = extractEntries(timeline, entry => entry.type === 'sentence')
15
+
16
+ const timelineWordEntryGroups: Timeline[] = []
17
+ const timelineWordGroups: string[][] = []
18
+
19
+ for (const sentenceEntry of timelineSentenceEntries) {
20
+ const wordEntryGroup = sentenceEntry.timeline!
21
+ .filter(wordEntry => isWord(wordEntry.text))
22
+
23
+ timelineWordEntryGroups.push(wordEntryGroup)
24
+ timelineWordGroups.push(wordEntryGroup.map(wordEntry => wordEntry.text))
25
+ }
26
+
27
+ const timelineWordEntriesFiltered = timelineWordEntryGroups.flat()
28
+
29
+ const textSentences = splitToSentences(text, textLangCode)
30
+
31
+ const textWordGroups: string[][] = []
32
+
33
+ for (const sentenceText of textSentences) {
34
+ let wordGroup = await splitToWords(sentenceText, textLangCode)
35
+ wordGroup = wordGroup.filter(word => isWord(word))
36
+
37
+ textWordGroups.push(wordGroup)
38
+ }
39
+
40
+ const textWords = textWordGroups.flat()
41
+
42
+ logger.end()
43
+
44
+ const wordMappingEntries = await alignWordsToWordsSemantically(timelineWordGroups, textWordGroups)
45
+
46
+ logger.start(`Build timeline for translation`)
47
+
48
+ const mappingGroups = new Map<number, number[]>()
49
+
50
+ for (const wordMappingEntry of wordMappingEntries) {
51
+ const wordIndex1 = wordMappingEntry.wordIndex1
52
+ const wordIndex2 = wordMappingEntry.wordIndex2
53
+
54
+ let group = mappingGroups.get(wordIndex1)
55
+
56
+ if (!group) {
57
+ group = []
58
+ mappingGroups.set(wordIndex1, group)
59
+ }
60
+
61
+ if (!group.includes(wordIndex2)) {
62
+ group.push(wordIndex2)
63
+ }
64
+ }
65
+
66
+ type TimeSlice = { startTime: number, endTime: number }
67
+
68
+ const timeSlicesLookup = new Map<number, TimeSlice[]>()
69
+
70
+ for (const [wordIndex1, mappedWordIndexes] of mappingGroups) {
71
+ if (mappedWordIndexes.length === 0) {
72
+ continue
73
+ }
74
+
75
+ const startTime = timelineWordEntriesFiltered[wordIndex1].startTime
76
+ const endTime = timelineWordEntriesFiltered[wordIndex1].endTime
77
+
78
+ const splitCount = mappedWordIndexes.length
79
+
80
+ const sliceDuration = (endTime - startTime) / splitCount
81
+
82
+ let timeOffset = 0
83
+
84
+ for (let i = 0; i < splitCount; i++) {
85
+ const timeSlice: TimeSlice = {
86
+ startTime: startTime + timeOffset,
87
+ endTime: startTime + timeOffset + sliceDuration
88
+ }
89
+
90
+ const wordIndex2 = mappedWordIndexes[i]
91
+
92
+ let timeSlicesForTargetWord = timeSlicesLookup.get(wordIndex2)
93
+
94
+ if (!timeSlicesForTargetWord) {
95
+ timeSlicesForTargetWord = []
96
+ timeSlicesLookup.set(wordIndex2, timeSlicesForTargetWord)
97
+ }
98
+
99
+ timeSlicesForTargetWord.push(timeSlice)
100
+
101
+ timeOffset += sliceDuration
102
+ }
103
+ }
104
+
105
+ const resultTimeline: Timeline = []
106
+
107
+ for (const [key, value] of timeSlicesLookup) {
108
+ resultTimeline.push({
109
+ type: 'word',
110
+ text: textWords[key],
111
+ startTime: value[0].startTime,
112
+ endTime: value[value.length - 1].endTime
113
+ })
114
+ }
115
+
116
+ logger.end()
117
+
118
+ return resultTimeline
119
+ }
120
+
121
+ export async function alignWordsToWordsSemantically(wordsGroups1: string[][], wordsGroups2: string[][], windowTokenCount = 1000 * 1000) {
122
+ const logger = new Logger()
123
+
124
+ // Load embedding model
125
+ const modelPath = await loadPackage(`xenova-multilingual-e5-small-fp16`)
126
+
127
+ const embeddingModel = new E5TextEmbedding(modelPath)
128
+
129
+ logger.start(`Initialize E5 embedding model`)
130
+ await embeddingModel.initializeIfNeeded()
131
+
132
+ async function extractEmbeddingsFromWordGroups(wordGroups: string[][]) {
133
+ const logger = new Logger()
134
+
135
+ const maxTokensPerFragment = 512
136
+ const { Tensor } = await import('@echogarden/transformers-nodejs-lite')
137
+
138
+ const words: string[] = []
139
+
140
+ const embeddings: TokenEmbeddingData[] = []
141
+ const tokenToWordIndexMapping: number[] = []
142
+
143
+ for (const wordGroup of wordGroups) {
144
+ const { joinedText: joinedTextForGroup, offsets: offsetsForGroup } = joinAndGetOffsets(wordGroup)
145
+
146
+ logger.start(`Tokenize text`)
147
+ const inputsForGroup = await embeddingModel.tokenizeToModelInputs(joinedTextForGroup)
148
+
149
+ logger.start(`Infer embeddings for text`)
150
+
151
+ const allTokenIds = inputsForGroup['input_ids'].data
152
+ const allAttentionMask = inputsForGroup['attention_mask'].data
153
+
154
+ let embeddingsForGroup: TokenEmbeddingData[] = []
155
+
156
+ for (let tokenStart = 0; tokenStart < allTokenIds.length; tokenStart += maxTokensPerFragment) {
157
+ const tokenEnd = Math.min(tokenStart + maxTokensPerFragment, allTokenIds.length)
158
+ const fragmentTokenCount = tokenEnd - tokenStart
159
+
160
+ const fragmentInputIdsTensor = new Tensor('int64', allTokenIds.slice(tokenStart, tokenEnd), [1, fragmentTokenCount])
161
+ const fragmentAttentionMaskTensor = new Tensor('int64', allAttentionMask.slice(tokenStart, tokenEnd), [1, fragmentTokenCount])
162
+
163
+ const inputsForFragment = { input_ids: fragmentInputIdsTensor, attention_mask: fragmentAttentionMaskTensor }
164
+
165
+ const embeddingsForFragment = await embeddingModel.inferTokenEmbeddings(inputsForFragment)
166
+
167
+ embeddingsForGroup.push(...embeddingsForFragment)
168
+ }
169
+
170
+ logger.start(`Compute token to word mapping for text`)
171
+ const filteredEmbeddingsForGroup = embeddingsForGroup.filter((embedding) => embedding.text !== '▁' && embedding.text !== '<s>' && embedding.text !== '</s>')
172
+ const tokenToWordIndexMappingForGroup = mapTokenEmbeddingsToWordIndexes(filteredEmbeddingsForGroup, joinedTextForGroup, offsetsForGroup)
173
+ const tokenToWordIndexMappingForGroupWithOffset = tokenToWordIndexMappingForGroup.map(value => words.length + value)
174
+
175
+ embeddings.push(...filteredEmbeddingsForGroup)
176
+ tokenToWordIndexMapping.push(...tokenToWordIndexMappingForGroupWithOffset)
177
+
178
+ words.push(...wordGroup)
179
+ }
180
+
181
+ return { words, embeddings, tokenToWordIndexMapping }
182
+ }
183
+
184
+ logger.start(`Extract embeddings from source 1`)
185
+ const {
186
+ words: words1,
187
+ embeddings: embeddings1,
188
+ tokenToWordIndexMapping: tokenToWordIndexMapping1
189
+ } = await extractEmbeddingsFromWordGroups(wordsGroups1)
190
+
191
+ logger.start(`Extract embeddings from source 2`)
192
+ const {
193
+ words: words2,
194
+ embeddings: embeddings2,
195
+ tokenToWordIndexMapping: tokenToWordIndexMapping2
196
+ } = await extractEmbeddingsFromWordGroups(wordsGroups2)
197
+
198
+ // Align
199
+ function costFunction(a: TokenEmbeddingData, b: TokenEmbeddingData) {
200
+ const aIsPunctuation = isPunctuation(a.text)
201
+ const bIsPunctuation = isPunctuation(b.text)
202
+
203
+ if (aIsPunctuation === bIsPunctuation) {
204
+ return cosineDistance(a.embeddingVector, b.embeddingVector)
205
+ } else {
206
+ return 1.0
207
+ }
208
+ }
209
+
210
+ logger.start(`Align token embedding vectors using DTW`)
211
+
212
+ const { path } = alignDTWWindowed(embeddings1, embeddings2, costFunction, windowTokenCount)
213
+
214
+ // Use alignment path to words to words
215
+ logger.start(`Map tokens to words`)
216
+
217
+ const wordMapping: WordMapping[] = []
218
+
219
+ for (let i = 0; i < path.length; i++) {
220
+ const pathEntry = path[i]
221
+
222
+ const sourceTokenIndex = pathEntry.source
223
+ const destTokenIndex = pathEntry.dest
224
+
225
+ const mappedWordIndex1 = tokenToWordIndexMapping1[sourceTokenIndex]
226
+ const mappedWordIndex2 = tokenToWordIndexMapping2[destTokenIndex]
227
+
228
+ wordMapping.push({
229
+ wordIndex1: mappedWordIndex1,
230
+ word1: words1[mappedWordIndex1],
231
+ wordIndex2: mappedWordIndex2,
232
+ word2: words2[mappedWordIndex2],
233
+ })
234
+ }
235
+
236
+ logger.end()
237
+
238
+ return wordMapping
239
+ }
240
+
241
+ function mapTokenEmbeddingsToWordIndexes(embeddings: TokenEmbeddingData[], text: string, textWordOffsets: number[]) {
242
+ const tokenToWordIndex: number[] = []
243
+
244
+ let currentTextOffset = 0
245
+
246
+ for (let i = 0; i < embeddings.length; i++) {
247
+ const embedding = embeddings[i]
248
+ let tokenText = embedding.text
249
+
250
+ if (tokenText === '<s>' || tokenText === '</s>') {
251
+ tokenToWordIndex.push(-1)
252
+
253
+ continue
254
+ }
255
+
256
+ if (tokenText.startsWith('▁')) {
257
+ tokenText = tokenText.substring(1)
258
+ }
259
+
260
+ const matchPosition = text.indexOf(tokenText, currentTextOffset)
261
+
262
+ if (matchPosition === -1) {
263
+ throw new Error(`Token '${tokenText}' not found in text`)
264
+ }
265
+
266
+ currentTextOffset = matchPosition + tokenText.length
267
+
268
+ let tokenMatchingWordIndex = textWordOffsets.findIndex((index) => index > matchPosition)
269
+
270
+ if (tokenMatchingWordIndex === -1) {
271
+ throw new Error(`Token '${tokenText}' not found in text`)
272
+ } else {
273
+ tokenMatchingWordIndex = Math.max(tokenMatchingWordIndex - 1, 0)
274
+ }
275
+
276
+ tokenToWordIndex.push(tokenMatchingWordIndex)
277
+ }
278
+
279
+ return tokenToWordIndex
280
+ }
281
+
282
+ function joinAndGetOffsets(words: string[]) {
283
+ let joinedText = ''
284
+ const offsets: number[] = []
285
+
286
+ let offset = 0
287
+
288
+ for (const word of words) {
289
+ const extendedWord = `${word} `
290
+ joinedText += extendedWord
291
+
292
+ offsets.push(offset)
293
+
294
+ offset += extendedWord.length
295
+ }
296
+
297
+ offsets.push(joinedText.length)
298
+
299
+ return { joinedText, offsets }
300
+ }
301
+
302
+ export class E5TextEmbedding {
303
+ tokenizer?: PreTrainedTokenizer
304
+ model?: PreTrainedModel
305
+
306
+ constructor(public readonly modelPath: string) {
307
+ }
308
+
309
+ async tokenizeToModelInputs(text: string) {
310
+ await this.initializeIfNeeded()
311
+
312
+ const inputs = await this.tokenizer!(text)
313
+
314
+ return inputs
315
+ }
316
+
317
+ async inferTokenEmbeddings(inputs: any) {
318
+ await this.initializeIfNeeded()
319
+
320
+ const tokensText = this.tokenizer!.model.convert_ids_to_tokens(Array.from(inputs.input_ids.data))
321
+
322
+ const result = await this.model!(inputs)
323
+
324
+ const lastHiddenState = result.last_hidden_state
325
+
326
+ const tokenCount = lastHiddenState.dims[1]
327
+ const embeddingSize = lastHiddenState.dims[2]
328
+
329
+ const tokenEmbeddings: TokenEmbeddingData[] = []
330
+
331
+ for (let i = 0; i < tokenCount; i++) {
332
+ const tokenEmbeddingVector = lastHiddenState.data.slice(i * embeddingSize, (i + 1) * embeddingSize)
333
+
334
+ const tokenId = Number(inputs.input_ids.data[i])
335
+ const tokenText = tokensText[i]
336
+
337
+ tokenEmbeddings.push({
338
+ id: tokenId,
339
+ text: tokenText,
340
+ embeddingVector: tokenEmbeddingVector
341
+ })
342
+ }
343
+
344
+ return tokenEmbeddings
345
+ }
346
+
347
+ async initializeIfNeeded() {
348
+ if (this.tokenizer && this.model) {
349
+ return
350
+ }
351
+
352
+ const { AutoTokenizer, AutoModel } = await import('@echogarden/transformers-nodejs-lite')
353
+
354
+ this.tokenizer = await AutoTokenizer.from_pretrained(this.modelPath)
355
+ this.model = await AutoModel.from_pretrained(this.modelPath)
356
+ }
357
+ }
358
+
359
+ export interface TokenEmbeddingData {
360
+ id: number
361
+ text: string
362
+ embeddingVector: Float32Array
363
+ }
364
+
365
+ export interface WordMapping {
366
+ wordIndex1: number
367
+ word1: string
368
+
369
+ wordIndex2: number
370
+ word2: string
371
+ }
372
+
373
+ export const e5SupportedLanguages: string[] = [
374
+ 'af', // Afrikaans
375
+ 'am', // Amharic
376
+ 'ar', // Arabic
377
+ 'as', // Assamese
378
+ 'az', // Azerbaijani
379
+ 'be', // Belarusian
380
+ 'bg', // Bulgarian
381
+ 'bn', // Bengali
382
+ 'br', // Breton
383
+ 'bs', // Bosnian
384
+ 'ca', // Catalan
385
+ 'cs', // Czech
386
+ 'cy', // Welsh
387
+ 'da', // Danish
388
+ 'de', // German
389
+ 'el', // Greek
390
+ 'en', // English
391
+ 'eo', // Esperanto
392
+ 'es', // Spanish
393
+ 'et', // Estonian
394
+ 'eu', // Basque
395
+ 'fa', // Persian
396
+ 'fi', // Finnish
397
+ 'fr', // French
398
+ 'fy', // Western Frisian
399
+ 'ga', // Irish
400
+ 'gd', // Scottish Gaelic
401
+ 'gl', // Galician
402
+ 'gu', // Gujarati
403
+ 'ha', // Hausa
404
+ 'he', // Hebrew
405
+ 'hi', // Hindi
406
+ 'hr', // Croatian
407
+ 'hu', // Hungarian
408
+ 'hy', // Armenian
409
+ 'id', // Indonesian
410
+ 'is', // Icelandic
411
+ 'it', // Italian
412
+ 'ja', // Japanese
413
+ 'jv', // Javanese
414
+ 'ka', // Georgian
415
+ 'kk', // Kazakh
416
+ 'km', // Khmer
417
+ 'kn', // Kannada
418
+ 'ko', // Korean
419
+ 'ku', // Kurdish
420
+ 'ky', // Kyrgyz
421
+ 'la', // Latin
422
+ 'lo', // Lao
423
+ 'lt', // Lithuanian
424
+ 'lv', // Latvian
425
+ 'mg', // Malagasy
426
+ 'mk', // Macedonian
427
+ 'ml', // Malayalam
428
+ 'mn', // Mongolian
429
+ 'mr', // Marathi
430
+ 'ms', // Malay
431
+ 'my', // Burmese
432
+ 'ne', // Nepali
433
+ 'nl', // Dutch
434
+ 'no', // Norwegian
435
+ 'om', // Oromo
436
+ 'or', // Oriya
437
+ 'pa', // Panjabi
438
+ 'pl', // Polish
439
+ 'ps', // Pashto
440
+ 'pt', // Portuguese
441
+ 'ro', // Romanian
442
+ 'ru', // Russian
443
+ 'sa', // Sanskrit
444
+ 'sd', // Sindhi
445
+ 'si', // Sinhala
446
+ 'sk', // Slovak
447
+ 'sl', // Slovenian
448
+ 'so', // Somali
449
+ 'sq', // Albanian
450
+ 'sr', // Serbian
451
+ 'su', // Sundanese
452
+ 'sv', // Swedish
453
+ 'sw', // Swahili
454
+ 'ta', // Tamil
455
+ 'te', // Telugu
456
+ 'th', // Thai
457
+ 'tl', // Tagalog
458
+ 'tr', // Turkish
459
+ 'ug', // Uyghur
460
+ 'uk', // Ukrainian
461
+ 'ur', // Urdu
462
+ 'uz', // Uzbek
463
+ 'vi', // Vietnamese
464
+ 'xh', // Xhosa
465
+ 'yi', // Yiddish
466
+ 'zh', // Chinese
467
+ ]