echogarden 0.0.1 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (295) hide show
  1. package/README.md +59 -1
  2. package/data/lexicons/heteronyms.json +674 -0
  3. package/data/schemas/options.json +1057 -0
  4. package/data/tables/lcid-table.json +7427 -0
  5. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +2 -0
  6. package/dist/alignment/DTWMfccSequenceAlignment.js +26 -0
  7. package/dist/alignment/DTWMfccSequenceAlignment.js.map +1 -0
  8. package/dist/alignment/DTWSequenceAlignment.d.ts +5 -0
  9. package/dist/alignment/DTWSequenceAlignment.js +99 -0
  10. package/dist/alignment/DTWSequenceAlignment.js.map +1 -0
  11. package/dist/alignment/DTWSequenceAlignmentWindowed.d.ts +5 -0
  12. package/dist/alignment/DTWSequenceAlignmentWindowed.js +161 -0
  13. package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -0
  14. package/dist/alignment/LevenshteinSequenceAlignment.d.ts +5 -0
  15. package/dist/alignment/LevenshteinSequenceAlignment.js +102 -0
  16. package/dist/alignment/LevenshteinSequenceAlignment.js.map +1 -0
  17. package/dist/alignment/SpeechAlignment.d.ts +27 -0
  18. package/dist/alignment/SpeechAlignment.js +266 -0
  19. package/dist/alignment/SpeechAlignment.js.map +1 -0
  20. package/dist/api/API.d.ts +8 -0
  21. package/dist/api/API.js +10 -0
  22. package/dist/api/API.js.map +1 -0
  23. package/dist/api/APIOptions.d.ts +12 -0
  24. package/dist/api/APIOptions.js +2 -0
  25. package/dist/api/APIOptions.js.map +1 -0
  26. package/dist/api/Alignment.d.ts +32 -0
  27. package/dist/api/Alignment.js +134 -0
  28. package/dist/api/Alignment.js.map +1 -0
  29. package/dist/api/Denoising.d.ts +14 -0
  30. package/dist/api/Denoising.js +66 -0
  31. package/dist/api/Denoising.js.map +1 -0
  32. package/dist/api/Globals.d.ts +1 -0
  33. package/dist/api/Globals.js +2 -0
  34. package/dist/api/Globals.js.map +1 -0
  35. package/dist/api/LanguageDetection.d.ts +49 -0
  36. package/dist/api/LanguageDetection.js +118 -0
  37. package/dist/api/LanguageDetection.js.map +1 -0
  38. package/dist/api/Recognition.d.ts +49 -0
  39. package/dist/api/Recognition.js +162 -0
  40. package/dist/api/Recognition.js.map +1 -0
  41. package/dist/api/Synthesis.d.ts +125 -0
  42. package/dist/api/Synthesis.js +892 -0
  43. package/dist/api/Synthesis.js.map +1 -0
  44. package/dist/api/Translation.d.ts +25 -0
  45. package/dist/api/Translation.js +69 -0
  46. package/dist/api/Translation.js.map +1 -0
  47. package/dist/api/Vad.d.ts +25 -0
  48. package/dist/api/Vad.js +88 -0
  49. package/dist/api/Vad.js.map +1 -0
  50. package/dist/audio/AudioBufferConversion.d.ts +15 -0
  51. package/dist/audio/AudioBufferConversion.js +228 -0
  52. package/dist/audio/AudioBufferConversion.js.map +1 -0
  53. package/dist/audio/AudioPlayer.d.ts +9 -0
  54. package/dist/audio/AudioPlayer.js +232 -0
  55. package/dist/audio/AudioPlayer.js.map +1 -0
  56. package/dist/audio/AudioRecorder.d.ts +3 -0
  57. package/dist/audio/AudioRecorder.js +68 -0
  58. package/dist/audio/AudioRecorder.js.map +1 -0
  59. package/dist/audio/AudioUtilities.d.ts +48 -0
  60. package/dist/audio/AudioUtilities.js +209 -0
  61. package/dist/audio/AudioUtilities.js.map +1 -0
  62. package/dist/audio/SoxPath.d.ts +1 -0
  63. package/dist/audio/SoxPath.js +19 -0
  64. package/dist/audio/SoxPath.js.map +1 -0
  65. package/dist/cli/CLI.d.ts +8 -0
  66. package/dist/cli/CLI.js +860 -0
  67. package/dist/cli/CLI.js.map +1 -0
  68. package/dist/cli/CLIConfigFile.d.ts +3 -0
  69. package/dist/cli/CLIConfigFile.js +62 -0
  70. package/dist/cli/CLIConfigFile.js.map +1 -0
  71. package/dist/cli/CLILauncher.d.ts +2 -0
  72. package/dist/cli/CLILauncher.js +22 -0
  73. package/dist/cli/CLILauncher.js.map +1 -0
  74. package/dist/cli/CLIOptionsSchema.d.ts +5 -0
  75. package/dist/cli/CLIOptionsSchema.js +36 -0
  76. package/dist/cli/CLIOptionsSchema.js.map +1 -0
  77. package/dist/cli/CLIParser.d.ts +6 -0
  78. package/dist/cli/CLIParser.js +34 -0
  79. package/dist/cli/CLIParser.js.map +1 -0
  80. package/dist/cli/CLIStarter.d.ts +1 -0
  81. package/dist/cli/CLIStarter.js +3 -0
  82. package/dist/cli/CLIStarter.js.map +1 -0
  83. package/dist/codecs/FFMpegTranscoder.d.ts +20 -0
  84. package/dist/codecs/FFMpegTranscoder.js +169 -0
  85. package/dist/codecs/FFMpegTranscoder.js.map +1 -0
  86. package/dist/codecs/TIMITCodec.d.ts +9 -0
  87. package/dist/codecs/TIMITCodec.js +14 -0
  88. package/dist/codecs/TIMITCodec.js.map +1 -0
  89. package/dist/codecs/WaveCodec.d.ts +19 -0
  90. package/dist/codecs/WaveCodec.js +208 -0
  91. package/dist/codecs/WaveCodec.js.map +1 -0
  92. package/dist/denoising/RNNoise.d.ts +6 -0
  93. package/dist/denoising/RNNoise.js +68 -0
  94. package/dist/denoising/RNNoise.js.map +1 -0
  95. package/dist/dsp/BiquadFilter.d.ts +26 -0
  96. package/dist/dsp/BiquadFilter.js +399 -0
  97. package/dist/dsp/BiquadFilter.js.map +1 -0
  98. package/dist/dsp/FFT.d.ts +9 -0
  99. package/dist/dsp/FFT.js +135 -0
  100. package/dist/dsp/FFT.js.map +1 -0
  101. package/dist/dsp/MFCC.d.ts +25 -0
  102. package/dist/dsp/MFCC.js +162 -0
  103. package/dist/dsp/MFCC.js.map +1 -0
  104. package/dist/dsp/MelSpectogram.d.ts +19 -0
  105. package/dist/dsp/MelSpectogram.js +102 -0
  106. package/dist/dsp/MelSpectogram.js.map +1 -0
  107. package/dist/dsp/Rubberband.d.ts +52 -0
  108. package/dist/dsp/Rubberband.js +186 -0
  109. package/dist/dsp/Rubberband.js.map +1 -0
  110. package/dist/dsp/Sonic.d.ts +2 -0
  111. package/dist/dsp/Sonic.js +39 -0
  112. package/dist/dsp/Sonic.js.map +1 -0
  113. package/dist/dsp/SpeexResampler.d.ts +3 -0
  114. package/dist/dsp/SpeexResampler.js +53 -0
  115. package/dist/dsp/SpeexResampler.js.map +1 -0
  116. package/dist/math/VectorMath.d.ts +70 -0
  117. package/dist/math/VectorMath.js +564 -0
  118. package/dist/math/VectorMath.js.map +1 -0
  119. package/dist/nlp/ChineseSegmentation.d.ts +1 -0
  120. package/dist/nlp/ChineseSegmentation.js +53 -0
  121. package/dist/nlp/ChineseSegmentation.js.map +1 -0
  122. package/dist/nlp/CompromiseNLP.d.ts +15 -0
  123. package/dist/nlp/CompromiseNLP.js +66 -0
  124. package/dist/nlp/CompromiseNLP.js.map +1 -0
  125. package/dist/nlp/EspeakPhonemizer.d.ts +4 -0
  126. package/dist/nlp/EspeakPhonemizer.js +133 -0
  127. package/dist/nlp/EspeakPhonemizer.js.map +1 -0
  128. package/dist/nlp/IPA.d.ts +19 -0
  129. package/dist/nlp/IPA.js +113 -0
  130. package/dist/nlp/IPA.js.map +1 -0
  131. package/dist/nlp/JapaneseSegmentation.d.ts +1 -0
  132. package/dist/nlp/JapaneseSegmentation.js +40 -0
  133. package/dist/nlp/JapaneseSegmentation.js.map +1 -0
  134. package/dist/nlp/Lexicon.d.ts +17 -0
  135. package/dist/nlp/Lexicon.js +6 -0
  136. package/dist/nlp/Lexicon.js.map +1 -0
  137. package/dist/nlp/PhoneConversion.d.ts +6 -0
  138. package/dist/nlp/PhoneConversion.js +467 -0
  139. package/dist/nlp/PhoneConversion.js.map +1 -0
  140. package/dist/nlp/Segmentation.d.ts +41 -0
  141. package/dist/nlp/Segmentation.js +158 -0
  142. package/dist/nlp/Segmentation.js.map +1 -0
  143. package/dist/nlp/TextNormalizer.d.ts +4 -0
  144. package/dist/nlp/TextNormalizer.js +69 -0
  145. package/dist/nlp/TextNormalizer.js.map +1 -0
  146. package/dist/recognition/AmazonTranscribeSTT.d.ts +6 -0
  147. package/dist/recognition/AmazonTranscribeSTT.js +79 -0
  148. package/dist/recognition/AmazonTranscribeSTT.js.map +1 -0
  149. package/dist/recognition/AzureCognitiveServicesSTT.d.ts +7 -0
  150. package/dist/recognition/AzureCognitiveServicesSTT.js +51 -0
  151. package/dist/recognition/AzureCognitiveServicesSTT.js.map +1 -0
  152. package/dist/recognition/GoogleCloudSTT.d.ts +7 -0
  153. package/dist/recognition/GoogleCloudSTT.js +66 -0
  154. package/dist/recognition/GoogleCloudSTT.js.map +1 -0
  155. package/dist/recognition/SileroSTT.d.ts +9 -0
  156. package/dist/recognition/SileroSTT.js +125 -0
  157. package/dist/recognition/SileroSTT.js.map +1 -0
  158. package/dist/recognition/VoskSTT.d.ts +10 -0
  159. package/dist/recognition/VoskSTT.js +78 -0
  160. package/dist/recognition/VoskSTT.js.map +1 -0
  161. package/dist/recognition/WhisperSTT.d.ts +69 -0
  162. package/dist/recognition/WhisperSTT.js +977 -0
  163. package/dist/recognition/WhisperSTT.js.map +1 -0
  164. package/dist/server/Server.d.ts +1 -0
  165. package/dist/server/Server.js +18 -0
  166. package/dist/server/Server.js.map +1 -0
  167. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +9 -0
  168. package/dist/speech-language-detection/SileroLanguageDetection.js +47 -0
  169. package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -0
  170. package/dist/subtitles/Subtitles.d.ts +25 -0
  171. package/dist/subtitles/Subtitles.js +286 -0
  172. package/dist/subtitles/Subtitles.js.map +1 -0
  173. package/dist/synthesis/AwsPollyTTS.d.ts +7 -0
  174. package/dist/synthesis/AwsPollyTTS.js +51 -0
  175. package/dist/synthesis/AwsPollyTTS.js.map +1 -0
  176. package/dist/synthesis/AzureCognitiveServicesTTS.d.ts +9 -0
  177. package/dist/synthesis/AzureCognitiveServicesTTS.js +103 -0
  178. package/dist/synthesis/AzureCognitiveServicesTTS.js.map +1 -0
  179. package/dist/synthesis/CoquiServerTTS.d.ts +6 -0
  180. package/dist/synthesis/CoquiServerTTS.js +22 -0
  181. package/dist/synthesis/CoquiServerTTS.js.map +1 -0
  182. package/dist/synthesis/ElevenLabsTTS.d.ts +8 -0
  183. package/dist/synthesis/ElevenLabsTTS.js +48 -0
  184. package/dist/synthesis/ElevenLabsTTS.js.map +1 -0
  185. package/dist/synthesis/EspeakTTS.d.ts +46 -0
  186. package/dist/synthesis/EspeakTTS.js +353 -0
  187. package/dist/synthesis/EspeakTTS.js.map +1 -0
  188. package/dist/synthesis/FliteTTS.d.ts +17 -0
  189. package/dist/synthesis/FliteTTS.js +326 -0
  190. package/dist/synthesis/FliteTTS.js.map +1 -0
  191. package/dist/synthesis/GoogleCloudTTS.d.ts +16 -0
  192. package/dist/synthesis/GoogleCloudTTS.js +72 -0
  193. package/dist/synthesis/GoogleCloudTTS.js.map +1 -0
  194. package/dist/synthesis/GoogleTranslateTTS.d.ts +13 -0
  195. package/dist/synthesis/GoogleTranslateTTS.js +177 -0
  196. package/dist/synthesis/GoogleTranslateTTS.js.map +1 -0
  197. package/dist/synthesis/MicrosoftEdgeTTS.d.ts +10 -0
  198. package/dist/synthesis/MicrosoftEdgeTTS.js +216 -0
  199. package/dist/synthesis/MicrosoftEdgeTTS.js.map +1 -0
  200. package/dist/synthesis/SamTTS.d.ts +4 -0
  201. package/dist/synthesis/SamTTS.js +21 -0
  202. package/dist/synthesis/SamTTS.js.map +1 -0
  203. package/dist/synthesis/SapiTTS.d.ts +9 -0
  204. package/dist/synthesis/SapiTTS.js +211 -0
  205. package/dist/synthesis/SapiTTS.js.map +1 -0
  206. package/dist/synthesis/StreamlabsPollyTTS.d.ts +12 -0
  207. package/dist/synthesis/StreamlabsPollyTTS.js +88 -0
  208. package/dist/synthesis/StreamlabsPollyTTS.js.map +1 -0
  209. package/dist/synthesis/SvoxPicoTTS.d.ts +11 -0
  210. package/dist/synthesis/SvoxPicoTTS.js +236 -0
  211. package/dist/synthesis/SvoxPicoTTS.js.map +1 -0
  212. package/dist/synthesis/VitsTTS.d.ts +27 -0
  213. package/dist/synthesis/VitsTTS.js +359 -0
  214. package/dist/synthesis/VitsTTS.js.map +1 -0
  215. package/dist/tests/Test.d.ts +1 -0
  216. package/dist/tests/Test.js +10 -0
  217. package/dist/tests/Test.js.map +1 -0
  218. package/dist/text-language-detection/FastTextLanguageDetection.d.ts +2 -0
  219. package/dist/text-language-detection/FastTextLanguageDetection.js +38 -0
  220. package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -0
  221. package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +2 -0
  222. package/dist/text-language-detection/TinyLDLanguageDetection.js +12 -0
  223. package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -0
  224. package/dist/utilities/BinaryArrayConversion.d.ts +15 -0
  225. package/dist/utilities/BinaryArrayConversion.js +115 -0
  226. package/dist/utilities/BinaryArrayConversion.js.map +1 -0
  227. package/dist/utilities/Compression.d.ts +4 -0
  228. package/dist/utilities/Compression.js +67 -0
  229. package/dist/utilities/Compression.js.map +1 -0
  230. package/dist/utilities/FileDownloader.d.ts +3 -0
  231. package/dist/utilities/FileDownloader.js +147 -0
  232. package/dist/utilities/FileDownloader.js.map +1 -0
  233. package/dist/utilities/FileSystem.d.ts +30 -0
  234. package/dist/utilities/FileSystem.js +141 -0
  235. package/dist/utilities/FileSystem.js.map +1 -0
  236. package/dist/utilities/Hashing.d.ts +10 -0
  237. package/dist/utilities/Hashing.js +169 -0
  238. package/dist/utilities/Hashing.js.map +1 -0
  239. package/dist/utilities/Locale.d.ts +9 -0
  240. package/dist/utilities/Locale.js +65 -0
  241. package/dist/utilities/Locale.js.map +1 -0
  242. package/dist/utilities/Logger.d.ts +11 -0
  243. package/dist/utilities/Logger.js +49 -0
  244. package/dist/utilities/Logger.js.map +1 -0
  245. package/dist/utilities/NdArrayUtilities.d.ts +3 -0
  246. package/dist/utilities/NdArrayUtilities.js +22 -0
  247. package/dist/utilities/NdArrayUtilities.js.map +1 -0
  248. package/dist/utilities/ObjectUtilities.d.ts +4 -0
  249. package/dist/utilities/ObjectUtilities.js +132 -0
  250. package/dist/utilities/ObjectUtilities.js.map +1 -0
  251. package/dist/utilities/OpenPromise.d.ts +6 -0
  252. package/dist/utilities/OpenPromise.js +12 -0
  253. package/dist/utilities/OpenPromise.js.map +1 -0
  254. package/dist/utilities/PackageManager.d.ts +4 -0
  255. package/dist/utilities/PackageManager.js +46 -0
  256. package/dist/utilities/PackageManager.js.map +1 -0
  257. package/dist/utilities/RandomGenerator.d.ts +35 -0
  258. package/dist/utilities/RandomGenerator.js +149 -0
  259. package/dist/utilities/RandomGenerator.js.map +1 -0
  260. package/dist/utilities/TarballMaker.d.ts +4 -0
  261. package/dist/utilities/TarballMaker.js +50 -0
  262. package/dist/utilities/TarballMaker.js.map +1 -0
  263. package/dist/utilities/Timeline.d.ts +20 -0
  264. package/dist/utilities/Timeline.js +110 -0
  265. package/dist/utilities/Timeline.js.map +1 -0
  266. package/dist/utilities/Timer.d.ts +13 -0
  267. package/dist/utilities/Timer.js +70 -0
  268. package/dist/utilities/Timer.js.map +1 -0
  269. package/dist/utilities/Utilities.d.ts +68 -0
  270. package/dist/utilities/Utilities.js +305 -0
  271. package/dist/utilities/Utilities.js.map +1 -0
  272. package/dist/utilities/WasmMemoryManager.d.ts +142 -0
  273. package/dist/utilities/WasmMemoryManager.js +407 -0
  274. package/dist/utilities/WasmMemoryManager.js.map +1 -0
  275. package/dist/utilities/WebReader.d.ts +1 -0
  276. package/dist/utilities/WebReader.js +47 -0
  277. package/dist/utilities/WebReader.js.map +1 -0
  278. package/dist/utilities/WikipediaReader.d.ts +1 -0
  279. package/dist/utilities/WikipediaReader.js +31 -0
  280. package/dist/utilities/WikipediaReader.js.map +1 -0
  281. package/dist/voice-activity-detection/SileroVAD.d.ts +13 -0
  282. package/dist/voice-activity-detection/SileroVAD.js +58 -0
  283. package/dist/voice-activity-detection/SileroVAD.js.map +1 -0
  284. package/dist/voice-activity-detection/WebRtcVAD.d.ts +3 -0
  285. package/dist/voice-activity-detection/WebRtcVAD.js +53 -0
  286. package/dist/voice-activity-detection/WebRtcVAD.js.map +1 -0
  287. package/docs/CLI.md +234 -0
  288. package/docs/Development.md +3 -0
  289. package/docs/Engines.md +80 -0
  290. package/docs/Licenses.md +37 -0
  291. package/docs/Options.md +192 -0
  292. package/docs/Roadmap.md +16 -0
  293. package/docs/Technical.md +72 -0
  294. package/package.json +119 -19
  295. package/cli.js +0 -3
@@ -0,0 +1,977 @@
1
+ import Onnx from 'onnxruntime-node';
2
+ import { Logger } from '../utilities/Logger.js';
3
+ import { computeMelSpectogramUsingFilterbanks } from "../dsp/MelSpectogram.js";
4
+ import { clip, delay, roundToDigits, splitFloat32Array } from '../utilities/Utilities.js';
5
+ import { indexOfMax, logSoftmax, logSumExp, meanOfVector, medianFilter, softMax, stdDeviationOfVector } from '../math/VectorMath.js';
6
+ import { splitToWords, wordCharacterPattern } from '../nlp/Segmentation.js';
7
+ import { alignDTWWindowed } from '../alignment/DTWSequenceAlignmentWindowed.js';
8
+ import { deepClone } from '../utilities/ObjectUtilities.js';
9
+ import { getRawAudioDuration } from '../audio/AudioUtilities.js';
10
+ import { readAndParseJsonFile, readFile } from '../utilities/FileSystem.js';
11
+ import path from 'path';
12
+ import { getShortLanguageCode, languageCodeToName } from '../utilities/Locale.js';
13
+ import { loadPackage } from '../utilities/PackageManager.js';
14
+ export async function recognize(sourceRawAudio, modelName, modelDir, tokenizerDir, task, sourceLanguage) {
15
+ if (sourceRawAudio.sampleRate != 16000) {
16
+ throw new Error("Source audio must have a sampling rate of 16000");
17
+ }
18
+ const whisper = new Whisper(modelName, modelDir, tokenizerDir);
19
+ await whisper.initialize();
20
+ const result = await whisper.recognize(sourceRawAudio, task, sourceLanguage);
21
+ return result;
22
+ }
23
+ export async function align(sourceRawAudio, referenceText, modelName, modelDir, tokenizerDir, language) {
24
+ if (sourceRawAudio.sampleRate != 16000) {
25
+ throw new Error("Source audio must have a sampling rate of 16000");
26
+ }
27
+ const whisper = new Whisper(modelName, modelDir, tokenizerDir);
28
+ await whisper.initialize();
29
+ const timeline = await whisper.align(sourceRawAudio, referenceText, language);
30
+ return timeline;
31
+ }
32
+ export async function detectLanguage(sourceRawAudio, modelName, modelDir, tokenizerDir) {
33
+ if (sourceRawAudio.sampleRate != 16000) {
34
+ throw new Error("Source audio must have a sampling rate of 16000");
35
+ }
36
+ const whisper = new Whisper(modelName, modelDir, tokenizerDir);
37
+ await whisper.initialize();
38
+ const audioFeatures = await whisper.encodeAudio(sourceRawAudio);
39
+ const results = await whisper.detectLanguage(audioFeatures);
40
+ return results;
41
+ }
42
+ export class Whisper {
43
+ modelName;
44
+ modelDir;
45
+ tokenizerDir;
46
+ isMultiligualModel;
47
+ audioEncoder;
48
+ textDecoder;
49
+ textToTokenLookup = new Map();
50
+ tokenToTextLookup = new Map();
51
+ merges = [];
52
+ onnxOptions = {
53
+ logSeverityLevel: 2,
54
+ executionProviders: ['cpu']
55
+ };
56
+ tokenConfig;
57
+ constructor(modelName, modelDir, tokenizerDir) {
58
+ this.modelDir = modelDir;
59
+ this.modelName = modelName;
60
+ this.tokenizerDir = tokenizerDir;
61
+ this.isMultiligualModel = isMultiligualModel(this.modelName);
62
+ if (this.isMultiligualModel) {
63
+ this.tokenConfig = {
64
+ sotToken: 50258,
65
+ sotPrevToken: 50361,
66
+ eotToken: 50257,
67
+ noSpeechToken: 50362,
68
+ noTimestampsToken: 50363,
69
+ timestampTokensStart: 50364,
70
+ suppressedTokens: [1, 2, 6, 7, 8, 9, 10, 12, 14, 25, 26, 27, 28, 29, 31, 58, 59, 60, 61, 62, 63, 90, 91, 92, 93, 359, 503, 522, 542, 873, 893, 902, 918, 922, 931, 1350, 1853, 1982, 2460, 2627, 3246, 3253, 3268, 3536, 3846, 3961, 4183, 4667, 6585, 6647, 7273, 9061, 9383, 10428, 10929, 11938, 12033, 12331, 12562, 13793, 14157, 14635, 15265, 15618, 16553, 16604, 18362, 18956, 20075, 21675, 22520, 26130, 26161, 26435, 28279, 29464, 31650, 32302, 32470, 36865, 42863, 47425, 49870, 50254, 50258, 50360, 50361, 50362]
71
+ };
72
+ }
73
+ else {
74
+ this.tokenConfig = {
75
+ sotToken: 50257,
76
+ sotPrevToken: 50360,
77
+ eotToken: 50256,
78
+ noSpeechToken: 50361,
79
+ noTimestampsToken: 50362,
80
+ timestampTokensStart: 50363,
81
+ suppressedTokens: [1, 2, 7, 8, 9, 10, 14, 25, 26, 27, 28, 29, 31, 58, 59, 60, 61, 62, 63, 90, 91, 92, 93, 357, 366, 438, 532, 685, 705, 796, 930, 1058, 1220, 1267, 1279, 1303, 1343, 1377, 1391, 1635, 1782, 1875, 2162, 2361, 2488, 3467, 4008, 4211, 4600, 4808, 5299, 5855, 6329, 7203, 9609, 9959, 10563, 10786, 11420, 11709, 11907, 13163, 13697, 13700, 14808, 15306, 16410, 16791, 17992, 19203, 19510, 20724, 22305, 22935, 27007, 30109, 30420, 33409, 34949, 40283, 40493, 40549, 47282, 49146, 50257, 50359, 50360, 50361]
82
+ };
83
+ }
84
+ }
85
+ async initialize() {
86
+ const logger = new Logger();
87
+ await logger.startAsync("Load tokenizer data");
88
+ const encoderFilePath = path.join(this.modelDir, "encoder.onnx");
89
+ const decoderFilePath = path.join(this.modelDir, "decoder.onnx");
90
+ const vocabFilePath = path.join(this.tokenizerDir, "vocab.json");
91
+ const mergesFilePath = path.join(this.tokenizerDir, "merges.txt");
92
+ const vocabObject = await readAndParseJsonFile(vocabFilePath);
93
+ function bpeEncodedStrToString(str) {
94
+ const decodedChars = [];
95
+ for (const char of str) {
96
+ const decodedChar = vocabCharacterSetLookup[char];
97
+ if (decodedChar == undefined) {
98
+ throw new Error(`Invalid char: '${char}'`);
99
+ }
100
+ decodedChars.push(decodedChar);
101
+ }
102
+ return Buffer.from(decodedChars).toString("utf-8");
103
+ }
104
+ for (const key in vocabObject) {
105
+ const value = vocabObject[key];
106
+ const decodedKey = bpeEncodedStrToString(key);
107
+ this.textToTokenLookup.set(decodedKey, value);
108
+ this.tokenToTextLookup.set(value, decodedKey);
109
+ }
110
+ const mergesFileRawLines = (await readFile(mergesFilePath, "utf8")).trim().split(/\r?\n/g);
111
+ const mergesFileRawEntries = mergesFileRawLines.map(line => line.trim().split(" "));
112
+ this.merges = mergesFileRawEntries.map(entry => [bpeEncodedStrToString(entry[0]), bpeEncodedStrToString(entry[1])]);
113
+ await logger.startAsync(`Create ONNX inference session for model '${this.modelName}'`);
114
+ this.audioEncoder = await Onnx.InferenceSession.create(encoderFilePath, this.onnxOptions);
115
+ this.textDecoder = await Onnx.InferenceSession.create(decoderFilePath, this.onnxOptions);
116
+ logger.end();
117
+ }
118
+ async recognize(rawAudio, task, language) {
119
+ const logger = new Logger();
120
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
121
+ const audioSamples = rawAudio.audioChannels[0];
122
+ const sampleRate = rawAudio.sampleRate;
123
+ const audioDuration = getRawAudioDuration(rawAudio);
124
+ const maxAudioSamples = sampleRate * 30;
125
+ let previousPartTokens = [];
126
+ let timeline = [];
127
+ for (let audioOffset = 0; audioOffset < audioSamples.length;) {
128
+ const segmentStartTime = audioOffset / sampleRate;
129
+ await logger.startAsync(`\nPrepare audio part at time position ${roundToDigits(segmentStartTime, 2)}`);
130
+ const audioPartSamples = audioSamples.slice(audioOffset, audioOffset + maxAudioSamples);
131
+ const audioPartRawAudio = { audioChannels: [audioPartSamples], sampleRate };
132
+ const audioPartDuration = getRawAudioDuration(audioPartRawAudio);
133
+ logger.end();
134
+ const audioPartFeatures = await this.encodeAudio(audioPartRawAudio);
135
+ const isFinalPart = audioOffset + maxAudioSamples > audioSamples.length;
136
+ let initialTokens = [];
137
+ if (previousPartTokens.length > 0) {
138
+ initialTokens = [this.tokenConfig.sotPrevToken, ...previousPartTokens];
139
+ }
140
+ initialTokens = [...initialTokens, ...this.getInitialTokens(language, task)];
141
+ logger.end();
142
+ let { decodedTokens: partTokens, crossAttentionQKs: partCrossAttentionQKs } = await this.decodeTokens(audioPartFeatures, initialTokens, audioPartDuration, isFinalPart);
143
+ const partTranscript = this.tokensToText(partTokens.slice(initialTokens.length));
144
+ logger.log(`Recognized part transcript: "${partTranscript}"`);
145
+ const lastToken = partTokens[partTokens.length - 1];
146
+ const lastTokenIsTimestamp = lastToken >= timestampTokensStart;
147
+ let audioEndOffset;
148
+ if (!isFinalPart && lastTokenIsTimestamp) {
149
+ const timePosition = (lastToken - timestampTokensStart) * 0.02;
150
+ audioEndOffset = audioOffset + Math.floor(timePosition * sampleRate);
151
+ }
152
+ else {
153
+ audioEndOffset = Math.min(audioOffset + maxAudioSamples, audioSamples.length);
154
+ }
155
+ const segmentEndTime = audioEndOffset / sampleRate;
156
+ const segmentFrameCount = Math.floor((segmentEndTime - segmentStartTime) / 0.02);
157
+ await logger.startAsync(`Extract timeline for part`);
158
+ if (partTokens.length != partCrossAttentionQKs.length) {
159
+ throw new Error("Unexpected: partTokens.length != partCrossAttentionQKs.length");
160
+ }
161
+ //partTokens = partTokens.filter(token => token < timestampTokensStart)
162
+ //partCrossAttentionQKs = await this.inferCrossAttentionQKs(partTokens, audioPartFeatures)
163
+ partTokens = partTokens.slice(initialTokens.length);
164
+ partCrossAttentionQKs = partCrossAttentionQKs.slice(initialTokens.length);
165
+ //await this.addWordsToTimeline(timeline, partTokens, audioPartRawAudio, partCrossAttentionQKs, initialAudioTimeOffset, audioPartSamples.length / sampleRate)
166
+ const alignmentPath = await this.findAlignmentPathFromQKs(partCrossAttentionQKs, partTokens, 0, segmentFrameCount); //, alignmentHeadsIndexes[this.modelName])
167
+ const partTimeline = await this.getWordTimelineFromAlignmentPath(alignmentPath, partTokens, segmentStartTime, segmentEndTime);
168
+ timeline.push(...partTimeline);
169
+ audioOffset = audioEndOffset;
170
+ previousPartTokens = partTokens.filter(token => token < this.tokenConfig.eotToken);
171
+ logger.end();
172
+ }
173
+ if (timeline.length > 0) {
174
+ timeline[timeline.length - 1].endTime = audioDuration;
175
+ }
176
+ timeline = this.mergeSuccessiveWordFragmentsInTimeline(timeline);
177
+ timeline.forEach(entry => { entry.text = entry.text.trim(); });
178
+ timeline = timeline.filter(entry => wordCharacterPattern.test(entry.text));
179
+ const transcript = timeline.map(entry => entry.text).join(" ");
180
+ logger.end();
181
+ return { transcript, timeline };
182
+ }
183
+ async align(rawAudio, referenceText, language) {
184
+ const logger = new Logger();
185
+ await logger.startAsync("Prepare for alignment");
186
+ const audioDuration = Math.min(getRawAudioDuration(rawAudio), 30);
187
+ const audioFrameCount = Math.floor(audioDuration / 0.02);
188
+ const initialTokens = this.getInitialTokens(language, "transcribe", true);
189
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
190
+ const eotToken = this.tokenConfig.eotToken;
191
+ let tokens = [...initialTokens, ...await this.textToTokens(referenceText, language), eotToken];
192
+ logger.end();
193
+ const audioFeatures = await this.encodeAudio(rawAudio);
194
+ await logger.startAsync("Infer cross-attention QKs");
195
+ let crossAttentionQKs = await this.inferCrossAttentionQKs(tokens, audioFeatures);
196
+ tokens = tokens.slice(initialTokens.length, tokens.length - 1);
197
+ crossAttentionQKs = crossAttentionQKs.slice(initialTokens.length, crossAttentionQKs.length - 1);
198
+ await logger.startAsync("Extract word timeline");
199
+ const alignmentPath = await this.findAlignmentPathFromQKs(crossAttentionQKs, tokens, 0, audioFrameCount); //, this.getAlignmentHeadIndexes())
200
+ let timeline = await this.getWordTimelineFromAlignmentPath(alignmentPath, tokens, 0, audioDuration);
201
+ timeline = this.mergeSuccessiveWordFragmentsInTimeline(timeline);
202
+ timeline.forEach(entry => { entry.text = entry.text.trim(); });
203
+ timeline = timeline.filter(entry => wordCharacterPattern.test(entry.text));
204
+ logger.end();
205
+ return timeline;
206
+ }
207
+ async detectLanguage(audioFeatures) {
208
+ const logger = new Logger();
209
+ if (!this.isMultiligualModel) {
210
+ throw new Error("Language detection only works for a multilingual model");
211
+ }
212
+ // Prepare and run decoder
213
+ logger.startAsync("Detect language with Whisper model");
214
+ const sotToken = this.tokenConfig.sotToken;
215
+ const initialTokens = [sotToken];
216
+ const offset = 0;
217
+ const initialKvDimensions = this.getKvDimensions(1, initialTokens.length);
218
+ const kvCacheTensor = new Onnx.Tensor('float32', new Float32Array(initialKvDimensions[0] * initialKvDimensions[1] * initialKvDimensions[2] * initialKvDimensions[3]), initialKvDimensions);
219
+ const tokensTensor = new Onnx.Tensor('int64', new BigInt64Array(initialTokens.map(token => BigInt(token))), [1, initialTokens.length]);
220
+ const offsetTensor = new Onnx.Tensor('int64', new BigInt64Array([BigInt(offset)]), []);
221
+ const decoderInputs = { tokens: tokensTensor, audio_features: audioFeatures, kv_cache: kvCacheTensor, offset: offsetTensor };
222
+ const decoderOutputs = await this.textDecoder.run(decoderInputs);
223
+ const logitsBuffer = decoderOutputs["logits"].data;
224
+ const languageTokensLogits = Array.from(logitsBuffer.slice(sotToken + 1, sotToken + 1 + 99));
225
+ const languageTokensProbabilities = softMax(languageTokensLogits, 1.0);
226
+ const results = [];
227
+ for (const language in languageIdLookup) {
228
+ const langId = languageIdLookup[language];
229
+ const probability = languageTokensProbabilities[langId];
230
+ results.push({
231
+ language,
232
+ languageName: languageCodeToName(language),
233
+ probability
234
+ });
235
+ }
236
+ results.sort((entry1, entry2) => entry2.probability - entry1.probability);
237
+ logger.end();
238
+ return results;
239
+ }
240
+ async decodeTokens(audioFeatures, initialTokens, audioDuration, isFinalPart) {
241
+ const logger = new Logger();
242
+ await logger.startAsync("Decode text tokens with Whisper decoder model");
243
+ const noSpeechThreshold = 0.6;
244
+ const blankToken = this.textToTokenLookup.get(" ");
245
+ const suppressedTokens = this.tokenConfig.suppressedTokens;
246
+ const sotToken = this.tokenConfig.sotToken;
247
+ const eotToken = this.tokenConfig.eotToken;
248
+ const noTimestampsToken = this.tokenConfig.noTimestampsToken;
249
+ const noSpeechToken = this.tokenConfig.noSpeechToken;
250
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
251
+ const maxDecodedTokenCount = 200;
252
+ let decodedTokens = initialTokens.slice();
253
+ const initialKvDimensions = this.getKvDimensions(1, decodedTokens.length);
254
+ let kvCacheTensor = new Onnx.Tensor('float32', new Float32Array(initialKvDimensions[0] * initialKvDimensions[1] * initialKvDimensions[2] * initialKvDimensions[3]), initialKvDimensions);
255
+ let decodedTokensTimestampLogits = [new Array(1501)];
256
+ let lastTimestampTokenIndex = -1;
257
+ let timestampsSeenCount = 0;
258
+ let decodedTokensCrossAttentionQKs = [];
259
+ for (let i = 0; i < decodedTokens.length; i++) {
260
+ decodedTokensCrossAttentionQKs.push(undefined);
261
+ }
262
+ // Start decoding loop
263
+ for (let decodedTokenCount = 0; decodedTokenCount < maxDecodedTokenCount; decodedTokenCount++) {
264
+ //logger.log(decodedTokenCount)
265
+ const isInitialState = decodedTokens.length == initialTokens.length;
266
+ const tokensToDecode = isInitialState ? decodedTokens : [decodedTokens[decodedTokens.length - 1]];
267
+ const offset = isInitialState ? 0 : decodedTokens.length;
268
+ if (!isInitialState) {
269
+ // Reshape KV Cache tensor
270
+ const dims = kvCacheTensor.dims;
271
+ const currentKvCacheGroups = splitFloat32Array(kvCacheTensor.data, dims[2] * dims[3]);
272
+ const reshapedKvCacheTensor = new Onnx.Tensor('float32', new Float32Array(dims[0] * dims[1] * (decodedTokens.length) * dims[3]), [dims[0], dims[1], decodedTokens.length, dims[3]]);
273
+ const reshapedKvCacheGroups = splitFloat32Array(reshapedKvCacheTensor.data, decodedTokens.length * dims[3]);
274
+ for (let i = 0; i < dims[0]; i++) {
275
+ reshapedKvCacheGroups[i].set(currentKvCacheGroups[i]);
276
+ }
277
+ kvCacheTensor = reshapedKvCacheTensor;
278
+ }
279
+ // Prepare and run decoder
280
+ const tokensTensor = new Onnx.Tensor('int64', new BigInt64Array(tokensToDecode.map(token => BigInt(token))), [1, tokensToDecode.length]);
281
+ const offsetTensor = new Onnx.Tensor('int64', new BigInt64Array([BigInt(offset)]), []);
282
+ const decoderInputs = { tokens: tokensTensor, audio_features: audioFeatures, kv_cache: kvCacheTensor, offset: offsetTensor };
283
+ const decoderOutputs = await this.textDecoder.run(decoderInputs);
284
+ const logitsBuffer = decoderOutputs["logits"].data;
285
+ kvCacheTensor = decoderOutputs["output_kv_cache"];
286
+ // Compute logits
287
+ const resultLogits = splitFloat32Array(logitsBuffer, logitsBuffer.length / decoderOutputs["logits"].dims[1]);
288
+ const tokenLogits = resultLogits[resultLogits.length - 1];
289
+ const tokenTimestampLogits = Array.from(tokenLogits.slice(timestampTokensStart));
290
+ // Suppress tokens
291
+ for (let logitIndex = 0; logitIndex < tokenLogits.length; logitIndex++) {
292
+ const isWrongTokenForInitialState = isInitialState && (logitIndex == blankToken || logitIndex == eotToken);
293
+ const isInSupressedList = suppressedTokens.includes(logitIndex);
294
+ const isNoTimestampsToken = logitIndex == noTimestampsToken;
295
+ const shouldSupressToken = isWrongTokenForInitialState || isInSupressedList || isNoTimestampsToken;
296
+ if (shouldSupressToken) {
297
+ tokenLogits[logitIndex] = -Infinity;
298
+ }
299
+ }
300
+ // Compute best token
301
+ const logProbs = logSoftmax(tokenLogits);
302
+ const textTokenLogProbs = logProbs.slice(0, timestampTokensStart);
303
+ const timestampTokenLogProbs = logProbs.slice(timestampTokensStart);
304
+ const indexOfMaxTextLogProb = indexOfMax(textTokenLogProbs);
305
+ const valueOfMaxTextLogProb = textTokenLogProbs[indexOfMaxTextLogProb];
306
+ const indexOfMaxTimestampLogProb = indexOfMax(timestampTokenLogProbs);
307
+ const logSumExpOfTimestampTokenLogProbs = logSumExp(timestampTokenLogProbs);
308
+ const isTimestampToken = logSumExpOfTimestampTokenLogProbs > valueOfMaxTextLogProb;
309
+ const previousTokenWasTimestamp = decodedTokens[decodedTokens.length - 1] >= timestampTokensStart;
310
+ const secondPreviousTokenWasTimestamp = decodedTokens.length < 2 || decodedTokens[decodedTokens.length - 2] >= timestampTokensStart;
311
+ if (isTimestampToken && !previousTokenWasTimestamp) {
312
+ timestampsSeenCount += 1;
313
+ }
314
+ // Add best token
315
+ function addToken(tokenToAdd, timestampLogits) {
316
+ decodedTokens.push(tokenToAdd);
317
+ decodedTokensTimestampLogits.push(timestampLogits);
318
+ decodedTokensCrossAttentionQKs.push(decoderOutputs["cross_attention_qks"]);
319
+ }
320
+ if (isTimestampToken || (previousTokenWasTimestamp && !secondPreviousTokenWasTimestamp)) {
321
+ if (previousTokenWasTimestamp) {
322
+ const previousToken = decodedTokens[decodedTokens.length - 1];
323
+ const previousTokenTimestampLogits = decodedTokensTimestampLogits[decodedTokensTimestampLogits.length - 1];
324
+ addToken(previousToken, previousTokenTimestampLogits);
325
+ lastTimestampTokenIndex = decodedTokens.length;
326
+ const previousTokenTimestamp = (previousToken - timestampTokensStart) * 0.02;
327
+ if (previousTokenTimestamp >= audioDuration) {
328
+ break;
329
+ }
330
+ }
331
+ else {
332
+ addToken(timestampTokensStart + indexOfMaxTimestampLogProb, tokenTimestampLogits);
333
+ }
334
+ }
335
+ else if (indexOfMaxTextLogProb == eotToken) {
336
+ break;
337
+ }
338
+ else {
339
+ addToken(indexOfMaxTextLogProb, tokenTimestampLogits);
340
+ }
341
+ await delay(0);
342
+ }
343
+ if (timestampsSeenCount >= 2 && !isFinalPart) {
344
+ decodedTokens = decodedTokens.slice(0, lastTimestampTokenIndex);
345
+ decodedTokensTimestampLogits = decodedTokensTimestampLogits.slice(0, lastTimestampTokenIndex);
346
+ decodedTokensCrossAttentionQKs = decodedTokensCrossAttentionQKs.slice(0, lastTimestampTokenIndex);
347
+ }
348
+ logger.end();
349
+ // Return the tokens
350
+ return { decodedTokens, decodedTokensTimestampLogits, crossAttentionQKs: decodedTokensCrossAttentionQKs };
351
+ }
352
+ async inferCrossAttentionQKs(tokens, audioFeatures) {
353
+ const offset = 0;
354
+ const tokensTensor = new Onnx.Tensor('int64', new BigInt64Array(tokens.map(token => BigInt(token))), [1, tokens.length]);
355
+ const offsetTensor = new Onnx.Tensor('int64', new BigInt64Array([BigInt(offset)]), []);
356
+ const initialKvDimensions = this.getKvDimensions(1, tokens.length);
357
+ const kvCacheTensor = new Onnx.Tensor('float32', new Float32Array(initialKvDimensions[0] * initialKvDimensions[1] * initialKvDimensions[2] * initialKvDimensions[3]), initialKvDimensions);
358
+ const decoderInputs = { tokens: tokensTensor, audio_features: audioFeatures, kv_cache: kvCacheTensor, offset: offsetTensor };
359
+ const decoderOutputs = await this.textDecoder.run(decoderInputs);
360
+ const crossAttentionQKsTensor = decoderOutputs["cross_attention_qks"];
361
+ const tensorShape = crossAttentionQKsTensor.dims.slice();
362
+ const ndarray = (await import('ndarray')).default;
363
+ let qkArray = ndarray(crossAttentionQKsTensor.data, crossAttentionQKsTensor.dims.slice());
364
+ qkArray = qkArray.transpose(3, 0, 1, 2, 4);
365
+ const tokenCrossAttentionQKsTensors = [];
366
+ for (let i0 = 0; i0 < qkArray.shape[0]; i0++) {
367
+ const dataForToken = [];
368
+ for (let i1 = 0; i1 < qkArray.shape[1]; i1++) {
369
+ for (let i2 = 0; i2 < qkArray.shape[2]; i2++) {
370
+ for (let i3 = 0; i3 < qkArray.shape[3]; i3++) {
371
+ for (let i4 = 0; i4 < qkArray.shape[4]; i4++) {
372
+ dataForToken.push(qkArray.get(i0, i1, i2, i3, i4));
373
+ }
374
+ }
375
+ }
376
+ }
377
+ const newTensorShape = tensorShape.slice();
378
+ newTensorShape[3] = 1;
379
+ const newTensor = new Onnx.Tensor('float32', dataForToken, newTensorShape);
380
+ tokenCrossAttentionQKsTensors.push(newTensor);
381
+ }
382
+ return tokenCrossAttentionQKsTensors;
383
+ }
384
+ async encodeAudio(rawAudio) {
385
+ const logger = new Logger();
386
+ const audioSamples = rawAudio.audioChannels[0];
387
+ const sampleRate = rawAudio.sampleRate;
388
+ const fftOrder = 400;
389
+ const hopLength = 160;
390
+ const filterbankCount = 80;
391
+ const maxAudioSamples = sampleRate * 30;
392
+ const maxAudioFrames = 3000;
393
+ await logger.startAsync("Extract mel spectogram from audio part");
394
+ const paddedAudioSamples = new Float32Array(maxAudioSamples);
395
+ paddedAudioSamples.set(audioSamples.subarray(0, maxAudioSamples), 0);
396
+ const rawAudioPart = { audioChannels: [paddedAudioSamples], sampleRate };
397
+ const { melSpectogram } = await computeMelSpectogramUsingFilterbanks(rawAudioPart, fftOrder, fftOrder, hopLength, filterbanks);
398
+ await logger.startAsync("Normalize mel spectogram");
399
+ const logMelSpectogram = melSpectogram.map(spectrum => spectrum.map(mel => Math.log10(Math.max(mel, 1e-10))));
400
+ let maxLogMel = -Infinity;
401
+ for (const spectrum of logMelSpectogram) {
402
+ for (const mel of spectrum) {
403
+ if (mel > maxLogMel) {
404
+ maxLogMel = mel;
405
+ }
406
+ }
407
+ }
408
+ const normalizedLogMelSpectogram = logMelSpectogram.map(spectrum => spectrum.map(logMel => (Math.max(logMel, maxLogMel - 8) + 4) / 4));
409
+ const flattenedNormalizedLogMelSpectogram = new Float32Array(maxAudioFrames * filterbankCount);
410
+ for (let i = 0; i < filterbankCount; i++) {
411
+ for (let j = 0; j < maxAudioFrames; j++) {
412
+ flattenedNormalizedLogMelSpectogram[(i * maxAudioFrames) + j] = normalizedLogMelSpectogram[j][i];
413
+ }
414
+ }
415
+ await logger.startAsync("Encode mel spectogram with Whisper encoder model");
416
+ const inputTensor = new Onnx.Tensor('float32', flattenedNormalizedLogMelSpectogram, [1, filterbankCount, maxAudioFrames]);
417
+ const encoderInputs = { mel: inputTensor };
418
+ const encoderOutputs = await this.audioEncoder.run(encoderInputs);
419
+ const encodedAudioFeatures = encoderOutputs["output"];
420
+ logger.end();
421
+ return encodedAudioFeatures;
422
+ }
423
+ addSegmentsToTimeline(timeline, tokens, initialTimeOffset, audioDuration) {
424
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
425
+ for (let i = 0; i < tokens.length; i++) {
426
+ const token = tokens[i];
427
+ if (token == this.tokenConfig.sotToken || token == this.tokenConfig.eotToken) {
428
+ continue;
429
+ }
430
+ const tokenIsTimestamp = token >= timestampTokensStart;
431
+ const previousTokenWasTimestamp = tokens.length > 1 && tokens[i - 1] >= timestampTokensStart;
432
+ if (tokenIsTimestamp) {
433
+ if (previousTokenWasTimestamp) {
434
+ continue;
435
+ }
436
+ let startTime = initialTimeOffset + (token - timestampTokensStart) * 0.02;
437
+ startTime = Math.min(startTime, audioDuration);
438
+ if (timeline.length > 0) {
439
+ timeline[timeline.length - 1].endTime = startTime;
440
+ }
441
+ timeline.push({
442
+ type: "segment",
443
+ text: "",
444
+ startTime,
445
+ endTime: -1,
446
+ });
447
+ }
448
+ else {
449
+ if (timeline.length == 0) {
450
+ timeline.push({
451
+ type: "segment",
452
+ text: "",
453
+ startTime: initialTimeOffset,
454
+ endTime: -1,
455
+ });
456
+ }
457
+ const tokenText = this.tokenToTextLookup.get(token) || "";
458
+ timeline[timeline.length - 1].text += tokenText;
459
+ }
460
+ }
461
+ }
462
+ async addWordsToTimeline(timeline, tokens, rawAudio, crossAttentionQKs, initialAudioTimeOffset, duration) {
463
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
464
+ let segmentStartTime = 0;
465
+ let segmentTokens = [];
466
+ let segmentCrossAttentionQKs = [];
467
+ for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex++) {
468
+ const token = tokens[tokenIndex];
469
+ const tokenCrossAttentionQKs = crossAttentionQKs[tokenIndex];
470
+ const segmentTokensWithoutTimestamps = segmentTokens.filter(token => token < this.tokenConfig.timestampTokensStart);
471
+ const isTimestamp = token >= timestampTokensStart;
472
+ if (isTimestamp || tokenIndex == tokens.length - 1) {
473
+ let tokenTime;
474
+ if (isTimestamp) {
475
+ tokenTime = (token - timestampTokensStart) * 0.02;
476
+ }
477
+ else {
478
+ tokenTime = duration;
479
+ }
480
+ if (segmentTokensWithoutTimestamps.length > 0) {
481
+ const segmentEndTime = tokenTime;
482
+ const segmentStartFrame = Math.floor(segmentStartTime / 0.02);
483
+ let segmentEndFrame = Math.floor(segmentEndTime / 0.02);
484
+ if (segmentStartFrame == segmentEndFrame) {
485
+ segmentEndFrame += 1;
486
+ }
487
+ const segmentFrameCount = segmentEndFrame - segmentStartFrame;
488
+ const reinferCrossAttentionQKs = true;
489
+ if (reinferCrossAttentionQKs) {
490
+ const initialTokens = this.getInitialTokens('en', 'transcribe');
491
+ const tokensToDecode = [...initialTokens, ...segmentTokensWithoutTimestamps];
492
+ //const segmentAudioFeaturesBuffer = audioFeatures.data.slice(segmentStartFrame * audioFeatures.dims[2], segmentEndFrame * audioFeatures.dims[2])
493
+ //const segmentAudioFeatures = new Onnx.Tensor('float32', segmentAudioFeaturesBuffer, [1, segmentFrameCount, audioFeatures.dims[2]])
494
+ const segmentAudioSamples = rawAudio.audioChannels[0].slice(Math.floor(segmentStartTime * rawAudio.sampleRate), Math.floor(segmentEndTime * rawAudio.sampleRate));
495
+ const segmentRawAudio = { audioChannels: [segmentAudioSamples], sampleRate: rawAudio.sampleRate };
496
+ const segmentAudioFeatures = await this.encodeAudio(segmentRawAudio);
497
+ const reinferredCrossAttentionQKs = await this.inferCrossAttentionQKs(tokensToDecode, segmentAudioFeatures);
498
+ reinferredCrossAttentionQKs.slice(initialTokens.length);
499
+ const alignmentPath = await this.findAlignmentPathFromQKs(reinferredCrossAttentionQKs, tokensToDecode, 0, segmentFrameCount); //, alignmentHeadsIndexes[modelName])
500
+ const wordTimeline = await this.getWordTimelineFromAlignmentPath(alignmentPath, segmentTokensWithoutTimestamps, initialAudioTimeOffset + segmentStartTime, initialAudioTimeOffset + segmentEndTime, 0.0);
501
+ timeline.push(...wordTimeline);
502
+ }
503
+ else {
504
+ const alignmentPath = await this.findAlignmentPathFromQKs(segmentCrossAttentionQKs, segmentTokens, segmentStartFrame, segmentEndFrame); //, alignmentHeadsIndexes[modelName])
505
+ const wordTimeline = await this.getWordTimelineFromAlignmentPath(alignmentPath, segmentTokens, initialAudioTimeOffset, initialAudioTimeOffset + segmentEndTime, 0.0);
506
+ timeline.push(...wordTimeline);
507
+ }
508
+ }
509
+ segmentStartTime = tokenTime;
510
+ segmentTokens = [];
511
+ segmentCrossAttentionQKs = [];
512
+ }
513
+ segmentTokens.push(token);
514
+ segmentCrossAttentionQKs.push(tokenCrossAttentionQKs);
515
+ }
516
+ }
517
+ mergeSuccessiveWordFragmentsInTimeline(timeline) {
518
+ const resultTimeline = [];
519
+ for (const entry of timeline) {
520
+ if (entry.type != "word") {
521
+ continue;
522
+ }
523
+ if (resultTimeline.length > 0 && !entry.text.startsWith(" ")) {
524
+ const lastEntry = resultTimeline[resultTimeline.length - 1];
525
+ lastEntry.text += entry.text;
526
+ lastEntry.endTime = entry.endTime;
527
+ }
528
+ else {
529
+ resultTimeline.push(deepClone(entry));
530
+ }
531
+ }
532
+ return resultTimeline;
533
+ }
534
+ async getWordTimelineFromAlignmentPath(alignmentPath, tokens, startTimeOffset, endTimeOffset, correctionAmount = 0.0) {
535
+ const wordTimeline = [];
536
+ for (let pathIndex = 0; pathIndex < alignmentPath.length; pathIndex++) {
537
+ if (pathIndex != 0 && alignmentPath[pathIndex].source == alignmentPath[pathIndex - 1].source) {
538
+ continue;
539
+ }
540
+ const tokenMappingEntry = alignmentPath[pathIndex];
541
+ const token = tokens[tokenMappingEntry.source];
542
+ const tokenText = this.tokenToTextLookup.get(token);
543
+ if (token >= this.tokenConfig.eotToken || !tokenText) {
544
+ continue;
545
+ }
546
+ let startTime = startTimeOffset + (tokenMappingEntry.dest * 0.02);
547
+ startTime = Math.max(startTime + correctionAmount, startTimeOffset);
548
+ if (wordTimeline.length > 0) {
549
+ wordTimeline[wordTimeline.length - 1].endTime = startTime;
550
+ }
551
+ wordTimeline.push({
552
+ type: "word",
553
+ text: tokenText,
554
+ startTime,
555
+ endTime: -1
556
+ });
557
+ }
558
+ if (wordTimeline.length > 0) {
559
+ wordTimeline[wordTimeline.length - 1].endTime = endTimeOffset;
560
+ }
561
+ return wordTimeline;
562
+ }
563
+ async findAlignmentPathFromQKs(qksTensors, tokens, segmentStartFrame, segmentEndFrame, headIndexes) {
564
+ const segmentFrameCount = segmentEndFrame - segmentStartFrame;
565
+ if (segmentFrameCount == 0) {
566
+ throw new Error("Segment has 0 frames");
567
+ }
568
+ const tokenCount = qksTensors.length;
569
+ const layerCount = qksTensors[0].dims[0];
570
+ const headCount = qksTensors[0].dims[2];
571
+ const frameCount = qksTensors[0].dims[4];
572
+ if (!headIndexes) {
573
+ headIndexes = [];
574
+ for (let i = 0; i < layerCount * headCount; i++) {
575
+ //for (let i = Math.floor(layerCount * headCount / 2); i < layerCount * headCount; i++) {
576
+ headIndexes.push(i);
577
+ }
578
+ }
579
+ // Load attention head weights from tensors
580
+ const attentionHeads = []; // [heads, tokens, frames]
581
+ for (const headIndex of headIndexes) {
582
+ const attentionHead = []; // [tokens, frames]
583
+ for (let tokenIndex = 0; tokenIndex < tokenCount; tokenIndex++) {
584
+ const bufferOffset = headIndex * frameCount;
585
+ const startIndexInBuffer = bufferOffset + segmentStartFrame;
586
+ const endIndexInBuffer = bufferOffset + segmentEndFrame;
587
+ const framesForHead = qksTensors[tokenIndex].data.slice(startIndexInBuffer, endIndexInBuffer);
588
+ attentionHead.push(Array.from(framesForHead));
589
+ }
590
+ attentionHeads.push(attentionHead);
591
+ }
592
+ const applySoftmax = true;
593
+ const normalize = true;
594
+ const applyMedianFilter = true;
595
+ const fixateTimestampTokens = false;
596
+ const softmaxTemperature = 1.0;
597
+ const medianFilterWidth = 7;
598
+ if (applySoftmax) {
599
+ // Apply softmax to each token's frames
600
+ for (const head of attentionHeads) {
601
+ for (let tokenIndex = 0; tokenIndex < tokenCount; tokenIndex++) {
602
+ head[tokenIndex] = softMax(head[tokenIndex], softmaxTemperature);
603
+ }
604
+ }
605
+ }
606
+ if (normalize) {
607
+ // Normalize all weights in each individual head
608
+ for (const head of attentionHeads) {
609
+ const allWeightsForHead = head.flatMap(tokenFrames => tokenFrames);
610
+ const meanOfAllWeights = meanOfVector(allWeightsForHead);
611
+ const stdDeviationOfAllWeights = stdDeviationOfVector(allWeightsForHead);
612
+ for (const tokenFrames of head) {
613
+ for (let frameIndex = 0; frameIndex < tokenFrames.length; frameIndex++) {
614
+ tokenFrames[frameIndex] = (tokenFrames[frameIndex] - meanOfAllWeights) / stdDeviationOfAllWeights;
615
+ }
616
+ }
617
+ }
618
+ }
619
+ if (applyMedianFilter) {
620
+ // Apply median filter to each token's frames
621
+ for (const head of attentionHeads) {
622
+ for (let tokenIndex = 0; tokenIndex < tokenCount; tokenIndex++) {
623
+ head[tokenIndex] = medianFilter(head[tokenIndex], medianFilterWidth);
624
+ }
625
+ }
626
+ }
627
+ // Compute the mean for all layers and heads
628
+ const frameMeansForToken = [];
629
+ for (let i = 0; i < tokenCount; i++) {
630
+ const frameMeans = new Array(segmentFrameCount);
631
+ frameMeansForToken.push(frameMeans);
632
+ }
633
+ for (let tokenIndex = 0; tokenIndex < tokenCount; tokenIndex++) {
634
+ for (let frameIndex = 0; frameIndex < segmentFrameCount; frameIndex++) {
635
+ let sum = 0;
636
+ for (const head of attentionHeads) {
637
+ sum += head[tokenIndex][frameIndex];
638
+ }
639
+ const frameMean = sum / attentionHeads.length;
640
+ frameMeansForToken[tokenIndex][frameIndex] = frameMean;
641
+ }
642
+ }
643
+ if (fixateTimestampTokens) {
644
+ const timestampTokensStart = this.tokenConfig.timestampTokensStart;
645
+ for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex++) {
646
+ if (tokens[tokenIndex] >= timestampTokensStart) {
647
+ let timestampFrame = tokens[tokenIndex] - timestampTokensStart;
648
+ timestampFrame = clip(timestampFrame, segmentStartFrame, segmentEndFrame - 1);
649
+ frameMeansForToken[tokenIndex][timestampFrame] = 100;
650
+ }
651
+ }
652
+ }
653
+ // Perform DTW
654
+ const tokenIndexes = [...Array(tokenCount).keys()];
655
+ const frameIndexes = [...Array(segmentFrameCount).keys()];
656
+ let { path } = await alignDTWWindowed(tokenIndexes, frameIndexes, (tokenIndex, frameIndex) => {
657
+ return -frameMeansForToken[tokenIndex][frameIndex];
658
+ }, 1000);
659
+ path = path.map(entry => ({ source: entry.source, dest: segmentStartFrame + entry.dest }));
660
+ return path;
661
+ }
662
+ getKvDimensions(groupCount, length) {
663
+ const modelName = this.modelName;
664
+ if (modelName == "tiny" || modelName == "tiny.en") {
665
+ return [8, groupCount, length, 384];
666
+ }
667
+ else if (modelName == "base" || modelName == "base.en") {
668
+ return [12, groupCount, length, 512];
669
+ }
670
+ else if (modelName == "small" || modelName == "small.en") {
671
+ return [24, groupCount, length, 768];
672
+ }
673
+ else if (modelName == "medium" || modelName == "medium.en") {
674
+ return [48, groupCount, length, 1024];
675
+ }
676
+ else if (modelName == "large" || modelName == "large-v1" || modelName == "large-v2") {
677
+ return [64, groupCount, length, 1280];
678
+ }
679
+ else {
680
+ throw new Error(`Unsupported model: ${modelName}`);
681
+ }
682
+ }
683
+ getInitialTokens(language, task, disableTimestamps = false) {
684
+ const sotToken = this.tokenConfig.sotToken;
685
+ let initialTokens;
686
+ if (this.isMultiligualModel) {
687
+ const languageToken = sotToken + 1 + languageIdLookup[language];
688
+ const translateTaskToken = 50358;
689
+ const transcribeTaskToken = 50359;
690
+ const taskToken = task == "transcribe" ? transcribeTaskToken : translateTaskToken;
691
+ initialTokens = [sotToken, languageToken, taskToken];
692
+ }
693
+ else {
694
+ initialTokens = [sotToken];
695
+ }
696
+ if (disableTimestamps) {
697
+ initialTokens.push(this.tokenConfig.noTimestampsToken);
698
+ }
699
+ return initialTokens;
700
+ }
701
+ getAlignmentHeadIndexes() {
702
+ return alignmentHeadsIndexes[this.modelName];
703
+ }
704
+ tokensToText(tokens) {
705
+ return tokens.map(token => this.tokenToTextLookup.get(token) || "").join("").trim();
706
+ }
707
+ async textToTokens(text, language) {
708
+ const resultTokens = [];
709
+ const words = (await splitToWords(text, language)).filter(w => w.trim().length > 0);
710
+ //words = words.filter(word => wordCharacterPattern.test(word))
711
+ for (let i = 1; i < words.length; i++) {
712
+ words[i] = ` ${words[i]}`;
713
+ }
714
+ const allResultingSubwords = [];
715
+ for (const word of words) {
716
+ const tokenForEntireWord = this.textToTokenLookup.get(word);
717
+ if (tokenForEntireWord) {
718
+ resultTokens.push(tokenForEntireWord);
719
+ allResultingSubwords.push([word]);
720
+ continue;
721
+ }
722
+ const subwords = word.split("");
723
+ for (const mergeRule of this.merges) {
724
+ for (let i = 0; i < subwords.length - 1; i++) {
725
+ const currentSubword = subwords[i];
726
+ const nextSubword = subwords[i + 1];
727
+ if (currentSubword == mergeRule[0] && nextSubword == mergeRule[1]) {
728
+ subwords.splice(i, 2, mergeRule[0] + mergeRule[1]);
729
+ }
730
+ }
731
+ }
732
+ for (const subword of subwords) {
733
+ const tokenForSubword = this.textToTokenLookup.get(subword);
734
+ if (!tokenForSubword) {
735
+ throw new Error(`Failed tokenizing the given text. The word '${word}' contains a subword '${subword}' which is not in the vocabulary.`);
736
+ }
737
+ resultTokens.push(tokenForSubword);
738
+ }
739
+ allResultingSubwords.push(subwords);
740
+ }
741
+ return resultTokens;
742
+ }
743
+ }
744
+ const filterbanks = [
745
+ /* 0 */ { startIndex: 1, weights: [0.02486259490251541,] },
746
+ /* 1 */ { startIndex: 1, weights: [0.001990821911022067, 0.022871771827340126,] },
747
+ /* 2 */ { startIndex: 2, weights: [0.003981643822044134, 0.02088095061480999,] },
748
+ /* 3 */ { startIndex: 3, weights: [0.0059724655002355576, 0.018890129402279854,] },
749
+ /* 4 */ { startIndex: 4, weights: [0.007963287644088268, 0.01689930632710457,] },
750
+ /* 5 */ { startIndex: 5, weights: [0.009954108856618404, 0.014908484183251858,] },
751
+ /* 6 */ { startIndex: 6, weights: [0.011944931000471115, 0.012917662039399147,] },
752
+ /* 7 */ { startIndex: 7, weights: [0.013935752213001251, 0.010926840826869011,] },
753
+ /* 8 */ { startIndex: 8, weights: [0.015926575288176537, 0.0089360186830163,] },
754
+ /* 9 */ { startIndex: 9, weights: [0.017917396500706673, 0.006945197004824877,] },
755
+ /* 10 */ { startIndex: 10, weights: [0.01990821771323681, 0.004954374860972166,] },
756
+ /* 11 */ { startIndex: 11, weights: [0.021899040788412094, 0.0029635531827807426,] },
757
+ /* 12 */ { startIndex: 12, weights: [0.02388986200094223, 0.0009727313299663365,] },
758
+ /* 13 */ { startIndex: 13, weights: [0.025880683213472366,] },
759
+ /* 14 */ { startIndex: 14, weights: [0.025835324078798294,] },
760
+ /* 15 */ { startIndex: 14, weights: [0.0010180906392633915, 0.023844502866268158,] },
761
+ /* 16 */ { startIndex: 15, weights: [0.003008912317454815, 0.021853681653738022,] },
762
+ /* 17 */ { startIndex: 16, weights: [0.004999734461307526, 0.019862858578562737,] },
763
+ /* 18 */ { startIndex: 17, weights: [0.006990555673837662, 0.0178720373660326,] },
764
+ /* 19 */ { startIndex: 18, weights: [0.008981377817690372, 0.015881216153502464,] },
765
+ /* 20 */ { startIndex: 19, weights: [0.010972199961543083, 0.013890394009649754,] },
766
+ /* 21 */ { startIndex: 20, weights: [0.01296302117407322, 0.011899571865797043,] },
767
+ /* 22 */ { startIndex: 21, weights: [0.01495384331792593, 0.009908749721944332,] },
768
+ /* 23 */ { startIndex: 22, weights: [0.01694466546177864, 0.007917927578091621,] },
769
+ /* 24 */ { startIndex: 23, weights: [0.018935488536953926, 0.005927106365561485,] },
770
+ /* 25 */ { startIndex: 24, weights: [0.020874010398983955, 0.004040425643324852,] },
771
+ /* 26 */ { startIndex: 25, weights: [0.022114217281341553, 0.0033186059445142746,] },
772
+ /* 27 */ { startIndex: 26, weights: [0.02173672430217266, 0.0036109676584601402,] },
773
+ /* 28 */ { startIndex: 27, weights: [0.020497702062129974, 0.004762193653732538,] },
774
+ /* 29 */ { startIndex: 28, weights: [0.018486659973859787, 0.006592618301510811,] },
775
+ /* 30 */ { startIndex: 29, weights: [0.01585603691637516, 0.00896277092397213,] },
776
+ /* 31 */ { startIndex: 30, weights: [0.012738768011331558, 0.011751330457627773,] },
777
+ /* 32 */ { startIndex: 31, weights: [0.009250369854271412, 0.014853144995868206,] },
778
+ /* 33 */ { startIndex: 32, weights: [0.005490840878337622, 0.018177473917603493, 0.0028155462350696325,] },
779
+ /* 34 */ { startIndex: 33, weights: [0.0015463664894923568, 0.01632951945066452, 0.007420188747346401,] },
780
+ /* 35 */ { startIndex: 35, weights: [0.011181050911545753, 0.012018864043056965,] },
781
+ /* 36 */ { startIndex: 36, weights: [0.006065350491553545, 0.016561277210712433, 0.004360878840088844,] },
782
+ /* 37 */ { startIndex: 37, weights: [0.0010297985281795263, 0.012770536355674267, 0.009707189165055752,] },
783
+ /* 38 */ { startIndex: 39, weights: [0.006986402906477451, 0.01485429983586073, 0.004391219466924667,] },
784
+ /* 39 */ { startIndex: 40, weights: [0.001418047584593296, 0.011486922390758991, 0.010089744813740253, 0.00040022286702878773,] },
785
+ /* 40 */ { startIndex: 42, weights: [0.005411104764789343, 0.014735566452145576, 0.006518189795315266,] },
786
+ /* 41 */ { startIndex: 44, weights: [0.00827841367572546, 0.012277561239898205, 0.00396781275048852,] },
787
+ /* 42 */ { startIndex: 45, weights: [0.002187808509916067, 0.010184479877352715, 0.00998187530785799, 0.0022864851634949446,] },
788
+ /* 43 */ { startIndex: 47, weights: [0.00386943481862545, 0.011274894699454308, 0.008466221392154694, 0.0013397691072896123,] },
789
+ /* 44 */ { startIndex: 49, weights: [0.004820294212549925, 0.011678251437842846, 0.007608682848513126, 0.0010091039584949613,] },
790
+ /* 45 */ { startIndex: 51, weights: [0.005156961735337973, 0.011507894843816757, 0.007301822770386934, 0.0011901655234396458,] },
791
+ /* 46 */ { startIndex: 53, weights: [0.004982104524970055, 0.010863498784601688, 0.007451189681887627, 0.001791381393559277,] },
792
+ /* 47 */ { startIndex: 55, weights: [0.004385921638458967, 0.009832492098212242, 0.007973956875503063, 0.002732589840888977,] },
793
+ /* 48 */ { startIndex: 57, weights: [0.0034474546555429697, 0.008491347543895245, 0.008797688409686089, 0.00394382793456316,] },
794
+ /* 49 */ { startIndex: 59, weights: [0.0022357646375894547, 0.0069067515432834625, 0.009859241545200348, 0.005364237818866968, 0.0008692338014952838,] },
795
+ /* 50 */ { startIndex: 61, weights: [0.0008110002381727099, 0.005136650986969471, 0.00946230161935091, 0.0069410777650773525, 0.0027783995028585196,] },
796
+ /* 51 */ { startIndex: 64, weights: [0.003231203882023692, 0.007237049750983715, 0.00862883497029543, 0.004773912951350212, 0.0009189908159896731,] },
797
+ /* 52 */ { startIndex: 66, weights: [0.001233637798577547, 0.0049433219246566296, 0.008653006516397, 0.006818502210080624, 0.003248583758249879,] },
798
+ /* 53 */ { startIndex: 69, weights: [0.0026164355222135782, 0.006051854696124792, 0.008880467154085636, 0.005574479699134827, 0.002268492942675948,] },
799
+ /* 54 */ { startIndex: 71, weights: [0.0002863667905330658, 0.003467798000201583, 0.0066492292098701, 0.00787146482616663, 0.004809896927326918, 0.0017483289120718837,] },
800
+ /* 55 */ { startIndex: 74, weights: [0.0009245910914614797, 0.0038708120118826628, 0.00681703258305788, 0.007283343467861414, 0.004448124207556248, 0.0016129047144204378,] },
801
+ /* 56 */ { startIndex: 77, weights: [0.0011703289346769452, 0.003898728871718049, 0.006627128925174475, 0.0070473202504217625, 0.004421714693307877, 0.0017961094854399562,] },
802
+ /* 57 */ { startIndex: 80, weights: [0.0010892992140725255, 0.003615982597693801, 0.006142666097730398, 0.007102936040610075, 0.004671447444707155, 0.002239959081634879,] },
803
+ /* 58 */ { startIndex: 83, weights: [0.0007392280967906117, 0.0030791081953793764, 0.005418988410383463, 0.007397185545414686, 0.005145462695509195, 0.002893739379942417, 0.0006420162972062826,] },
804
+ /* 59 */ { startIndex: 86, weights: [0.00017068670422304422, 0.0023375742603093386, 0.004504461772739887, 0.0066713495180010796, 0.005798479542136192, 0.003713231300935149, 0.0016279831761494279,] },
805
+ /* 60 */ { startIndex: 90, weights: [0.0014345343224704266, 0.0034412189852446318, 0.005447904113680124, 0.006591092795133591, 0.004660011734813452, 0.002728930441662669, 0.0007978491485118866,] },
806
+ /* 61 */ { startIndex: 93, weights: [0.0004075043834745884, 0.002265830524265766, 0.004124156199395657, 0.005982482805848122, 0.005700822453945875, 0.003912510350346565, 0.0021241982467472553, 0.0003358862304594368,] },
807
+ /* 62 */ { startIndex: 97, weights: [0.0010099108330905437, 0.002730846870690584, 0.004451782442629337, 0.006172718480229378, 0.005150905344635248, 0.0034948070533573627, 0.0018387088784947991, 0.0001826105872169137,] },
808
+ /* 63 */ { startIndex: 101, weights: [0.0012943691108375788, 0.002888072282075882, 0.004481775686144829, 0.006075479090213776, 0.0048866597935557365, 0.003353001084178686, 0.0018193417927250266, 0.00028568264679051936,] },
809
+ /* 64 */ { startIndex: 105, weights: [0.0013131388695910573, 0.0027890161145478487, 0.004264893010258675, 0.0057407706044614315, 0.004859979264438152, 0.0034397069830447435, 0.0020194342359900475, 0.0005991620710119605,] },
810
+ /* 65 */ { startIndex: 109, weights: [0.0011121684219688177, 0.002478930866345763, 0.0038456933107227087, 0.0052124555222690105, 0.005028639920055866, 0.0037133716978132725, 0.002398103242740035, 0.0010828346712514758,] },
811
+ /* 66 */ { startIndex: 113, weights: [0.0007317548734135926, 0.0019974694587290287, 0.003263183869421482, 0.004528898745775223, 0.005355686880648136, 0.004137659445405006, 0.0029196315445005894, 0.0017016039928421378, 0.0004835762665607035,] },
812
+ /* 67 */ { startIndex: 117, weights: [0.00020713974663522094, 0.0013792773243039846, 0.0025514145381748676, 0.003723552217707038, 0.004895689897239208, 0.004680895246565342, 0.0035529187880456448, 0.0024249425623565912, 0.0012969663366675377, 0.00016899015463422984,] },
813
+ /* 68 */ { startIndex: 122, weights: [0.0006545265205204487, 0.0017400053329765797, 0.0028254841454327106, 0.003910962492227554, 0.004996441304683685, 0.0042709787376224995, 0.003226396394893527, 0.002181813819333911, 0.0011372314766049385, 9.264905384043232e-05,] },
814
+ /* 69 */ { startIndex: 127, weights: [0.000854626705404371, 0.001859853626228869, 0.002865080488845706, 0.003870307235047221, 0.00487553421407938, 0.00408313749358058, 0.003115783678367734, 0.0021484296303242445, 0.001181075582280755, 0.0002137213887181133,] },
815
+ /* 70 */ { startIndex: 132, weights: [0.0008483415003865957, 0.0017792496364563704, 0.0027101580053567886, 0.0036410661414265633, 0.004571974277496338, 0.004079728852957487, 0.003183893393725157, 0.002288057701662183, 0.0013922222424298525, 0.0004963868414051831,] },
816
+ /* 71 */ { startIndex: 137, weights: [0.0006716204807162285, 0.0015337044605985284, 0.002395788673311472, 0.0032578727696090937, 0.004119956865906715, 0.004227725323289633, 0.0033981208689510822, 0.0025685166474431753, 0.0017389123095199466, 0.0009093079133890569, 7.970355363795534e-05,] },
817
+ /* 72 */ { startIndex: 142, weights: [0.0003559796023182571, 0.0011543278815224767, 0.0019526762189343572, 0.002751024439930916, 0.0035493727773427963, 0.004347721114754677, 0.0037299629766494036, 0.002961693098768592, 0.00219342322088778, 0.0014251532265916467, 0.0006568834069184959,] },
818
+ /* 73 */ { startIndex: 148, weights: [0.0006682946113869548, 0.0014076193328946829, 0.0021469437051564455, 0.002886268775910139, 0.0036255933810025454, 0.004154576454311609, 0.0034431067761033773, 0.0027316368650645018, 0.0020201667211949825, 0.0013086966937407851, 0.0005972267827019095,] },
819
+ /* 74 */ { startIndex: 153, weights: [9.926508937496692e-05, 0.0007839298341423273, 0.001468594535253942, 0.0021532592363655567, 0.0028379240538924932, 0.0035225888714194298, 0.0039915177039802074, 0.0033326479606330395, 0.002673778682947159, 0.002014909405261278, 0.0013560398947447538, 0.0006971705006435513, 3.8301113818306476e-05,] },
820
+ /* 75 */ { startIndex: 159, weights: [0.00010181095422012731, 0.0007358568836934865, 0.0013699028640985489, 0.0020039486698806286, 0.002637994708493352, 0.0032720407471060753, 0.003906086552888155, 0.0033682563807815313, 0.0027580985333770514, 0.002147940918803215, 0.0015377833042293787, 0.0009276255150325596, 0.000317467754939571,] },
821
+ /* 76 */ { startIndex: 166, weights: [0.0005530364578589797, 0.0011402058880776167, 0.0017273754347115755, 0.0023145449813455343, 0.002901714527979493, 0.003488884074613452, 0.003523340215906501, 0.002958292607218027, 0.002393245231360197, 0.0018281979719176888, 0.001263150479644537, 0.0006981031037867069, 0.0001330557424807921,] },
822
+ /* 77 */ { startIndex: 172, weights: [0.0002608386566862464, 0.0008045974536798894, 0.0013483562506735325, 0.0018921148730441928, 0.0024358737282454967, 0.002979632467031479, 0.003523391205817461, 0.003251380519941449, 0.0027281083166599274, 0.002204835880547762, 0.001681563793681562, 0.001158291706815362, 0.0006350195035338402, 0.00011174729297636077,] },
823
+ /* 78 */ { startIndex: 179, weights: [0.0003849811910185963, 0.0008885387214832008, 0.001392096164636314, 0.0018956535495817661, 0.00239921105094254, 0.002902768552303314, 0.0034063260536640882, 0.003132763085886836, 0.0026481777895241976, 0.0021635922603309155, 0.0016790067311376333, 0.0011944210855290294, 0.0007098356145434082, 0.00022525011445395648,] },
824
+ /* 79 */ { startIndex: 186, weights: [0.000366741674952209, 0.0008330700220540166, 0.0012993983691558242, 0.0017657268326729536, 0.0022320549469441175, 0.002698383294045925, 0.0031647118739783764, 0.003141313325613737, 0.002692554146051407, 0.0022437951993197203, 0.00179503601975739, 0.0013462770730257034, 0.000897518009878695, 0.0004487590049393475,] },
825
+ ];
826
+ export async function loadPackagesAndGetPaths(modelName, languageCode) {
827
+ if (!modelName) {
828
+ if (languageCode) {
829
+ const shortLanguageCode = getShortLanguageCode(languageCode);
830
+ modelName = shortLanguageCode == "en" ? "tiny.en" : "tiny";
831
+ }
832
+ else {
833
+ modelName = "tiny";
834
+ }
835
+ }
836
+ const packageName = modelNameToPackageName[modelName];
837
+ const modelDir = await loadPackage(packageName);
838
+ const tokenizerPackagePath = await loadPackage(tokenizerPackageName);
839
+ const tokenizerDir = isMultiligualModel(modelName) ? path.join(tokenizerPackagePath, "multilingual") : path.join(tokenizerPackagePath, "gpt2");
840
+ return { modelName, modelDir, tokenizerDir };
841
+ }
842
+ export function isMultiligualModel(modelName) {
843
+ return !modelName.endsWith(".en");
844
+ }
845
+ export const modelNameToPackageName = {
846
+ "tiny": "whisper-tiny",
847
+ "tiny.en": "whisper-tiny.en",
848
+ "base": "whisper-base",
849
+ "base.en": "whisper-base.en",
850
+ "small": "whisper-small",
851
+ "small.en": "whisper-small.en",
852
+ "medium": "whisper-medium",
853
+ "medium.en": "whisper-medium.en",
854
+ "large": "whisper-large-v2",
855
+ "large-v1": "whisper-large-v1",
856
+ "large-v2": "whisper-large-v2"
857
+ };
858
+ export const tokenizerPackageName = "whisper-tokenizer";
859
+ const vocabCharacterSetLookup = {
860
+ "!": 33, "\"": 34, "#": 35, "$": 36, "%": 37, "&": 38, "'": 39, "(": 40, ")": 41, "*": 42, "+": 43, ",": 44, "-": 45, ".": 46, "/": 47, "0": 48, "1": 49, "2": 50, "3": 51, "4": 52, "5": 53, "6": 54,
861
+ "7": 55, "8": 56, "9": 57, ":": 58, ";": 59, "<": 60, "=": 61, ">": 62, "?": 63, "@": 64, "A": 65, "B": 66, "C": 67, "D": 68, "E": 69, "F": 70, "G": 71, "H": 72, "I": 73, "J": 74, "K": 75, "L": 76, "M": 77, "N": 78, "O": 79, "P": 80, "Q": 81, "R": 82, "S": 83, "T": 84, "U": 85, "V": 86, "W": 87, "X": 88, "Y": 89, "Z": 90, "[": 91, "\\": 92, "]": 93, "^": 94, "_": 95, "`": 96, "a": 97, "b": 98, "c": 99, "d": 100, "e": 101, "f": 102, "g": 103, "h": 104, "i": 105, "j": 106, "k": 107, "l": 108, "m": 109, "n": 110, "o": 111, "p": 112, "q": 113, "r": 114, "s": 115, "t": 116, "u": 117, "v": 118, "w": 119, "x": 120, "y": 121, "z": 122, "{": 123, "|": 124, "}": 125, "~": 126, "¡": 161, "¢": 162, "£": 163, "¤": 164, "¥": 165, "¦": 166, "§": 167, "¨": 168, "©": 169, "ª": 170, "«": 171, "¬": 172, "®": 174, "¯": 175, "°": 176, "±": 177, "²": 178, "³": 179, "´": 180, "µ": 181, "¶": 182, "·": 183, "¸": 184, "¹": 185, "º": 186, "»": 187, "¼": 188, "½": 189, "¾": 190, "¿": 191, "À": 192, "Á": 193, "Â": 194, "Ã": 195, "Ä": 196, "Å": 197, "Æ": 198, "Ç": 199, "È": 200, "É": 201, "Ê": 202, "Ë": 203, "Ì": 204, "Í": 205, "Î": 206, "Ï": 207, "Ð": 208, "Ñ": 209, "Ò": 210, "Ó": 211, "Ô": 212, "Õ": 213, "Ö": 214, "×": 215, "Ø": 216, "Ù": 217, "Ú": 218, "Û": 219, "Ü": 220, "Ý": 221, "Þ": 222, "ß": 223, "à": 224, "á": 225, "â": 226, "ã": 227, "ä": 228, "å": 229, "æ": 230, "ç": 231, "è": 232, "é": 233, "ê": 234, "ë": 235, "ì": 236, "í": 237, "î": 238, "ï": 239, "ð": 240, "ñ": 241, "ò": 242, "ó": 243, "ô": 244, "õ": 245, "ö": 246, "÷": 247, "ø": 248, "ù": 249, "ú": 250, "û": 251, "ü": 252, "ý": 253, "þ": 254, "ÿ": 255, "Ā": 0, "ā": 1, "Ă": 2, "ă": 3, "Ą": 4, "ą": 5, "Ć": 6, "ć": 7, "Ĉ": 8, "ĉ": 9, "Ċ": 10, "ċ": 11, "Č": 12, "č": 13, "Ď": 14, "ď": 15, "Đ": 16, "đ": 17, "Ē": 18, "ē": 19, "Ĕ": 20, "ĕ": 21, "Ė": 22, "ė": 23, "Ę": 24, "ę": 25, "Ě": 26, "ě": 27, "Ĝ": 28, "ĝ": 29, "Ğ": 30, "ğ": 31, "Ġ": 32, "ġ": 127, "Ģ": 128, "ģ": 129, "Ĥ": 130, "ĥ": 131, "Ħ": 132, "ħ": 133, "Ĩ": 134, "ĩ": 135, "Ī": 136, "ī": 137, "Ĭ": 138, "ĭ": 139, "Į": 140, "į": 141, "İ": 142, "ı": 143, "IJ": 144, "ij": 145, "Ĵ": 146, "ĵ": 147, "Ķ": 148, "ķ": 149, "ĸ": 150, "Ĺ": 151, "ĺ": 152, "Ļ": 153, "ļ": 154, "Ľ": 155, "ľ": 156, "Ŀ": 157, "ŀ": 158, "Ł": 159, "ł": 160, "Ń": 173
862
+ };
863
+ const languageIdLookup = {
864
+ "en": 0,
865
+ "zh": 1,
866
+ "de": 2,
867
+ "es": 3,
868
+ "ru": 4,
869
+ "ko": 5,
870
+ "fr": 6,
871
+ "ja": 7,
872
+ "pt": 8,
873
+ "tr": 9,
874
+ "pl": 10,
875
+ "ca": 11,
876
+ "nl": 12,
877
+ "ar": 13,
878
+ "sv": 14,
879
+ "it": 15,
880
+ "id": 16,
881
+ "hi": 17,
882
+ "fi": 18,
883
+ "vi": 19,
884
+ "iw": 20,
885
+ "uk": 21,
886
+ "el": 22,
887
+ "ms": 23,
888
+ "cs": 24,
889
+ "ro": 25,
890
+ "da": 26,
891
+ "hu": 27,
892
+ "ta": 28,
893
+ "no": 29,
894
+ "th": 30,
895
+ "ur": 31,
896
+ "hr": 32,
897
+ "bg": 33,
898
+ "lt": 34,
899
+ "la": 35,
900
+ "mi": 36,
901
+ "ml": 37,
902
+ "cy": 38,
903
+ "sk": 39,
904
+ "te": 40,
905
+ "fa": 41,
906
+ "lv": 42,
907
+ "bn": 43,
908
+ "sr": 44,
909
+ "az": 45,
910
+ "sl": 46,
911
+ "kn": 47,
912
+ "et": 48,
913
+ "mk": 49,
914
+ "br": 50,
915
+ "eu": 51,
916
+ "is": 52,
917
+ "hy": 53,
918
+ "ne": 54,
919
+ "mn": 55,
920
+ "bs": 56,
921
+ "kk": 57,
922
+ "sq": 58,
923
+ "sw": 59,
924
+ "gl": 60,
925
+ "mr": 61,
926
+ "pa": 62,
927
+ "si": 63,
928
+ "km": 64,
929
+ "sn": 65,
930
+ "yo": 66,
931
+ "so": 67,
932
+ "af": 68,
933
+ "oc": 69,
934
+ "ka": 70,
935
+ "be": 71,
936
+ "tg": 72,
937
+ "sd": 73,
938
+ "gu": 74,
939
+ "am": 75,
940
+ "yi": 76,
941
+ "lo": 77,
942
+ "uz": 78,
943
+ "fo": 79,
944
+ "ht": 80,
945
+ "ps": 81,
946
+ "tk": 82,
947
+ "nn": 83,
948
+ "mt": 84,
949
+ "sa": 85,
950
+ "lb": 86,
951
+ "my": 87,
952
+ "bo": 88,
953
+ "tl": 89,
954
+ "mg": 90,
955
+ "as": 91,
956
+ "tt": 92,
957
+ "haw": 93,
958
+ "ln": 94,
959
+ "ha": 95,
960
+ "ba": 96,
961
+ "jw": 97,
962
+ "su": 98,
963
+ };
964
+ const alignmentHeadsIndexes = {
965
+ "tiny.en": [6, 12, 17, 18, 19, 20, 21, 22],
966
+ "tiny": [14, 18, 20, 21, 22, 23],
967
+ "base.en": [27, 39, 41, 45, 47],
968
+ "base": [25, 34, 35, 39, 41, 42, 44, 46],
969
+ "small.en": [78, 84, 87, 92, 98, 101, 103, 108, 112, 116, 118, 120, 121, 122, 123, 126, 131, 134, 136],
970
+ "small": [63, 69, 96, 100, 103, 104, 108, 115, 117, 125],
971
+ "medium.en": [180, 225, 236, 238, 244, 256, 260, 265, 284, 286, 295, 298, 303, 320, 323, 329, 334, 348],
972
+ "medium": [223, 244, 255, 257, 320, 372],
973
+ "large-v1": [199, 222, 224, 237, 447, 451, 457, 462, 475],
974
+ "large-v2": [212, 277, 331, 332, 333, 355, 356, 364, 371, 379, 391, 422, 423, 443, 449, 452, 465, 467, 473, 505, 521, 532, 555],
975
+ "large": [212, 277, 331, 332, 333, 355, 356, 364, 371, 379, 391, 422, 423, 443, 449, 452, 465, 467, 473, 505, 521, 532, 555],
976
+ };
977
+ //# sourceMappingURL=WhisperSTT.js.map