echogarden 1.4.4 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (240) hide show
  1. package/data/schemas/options.json +310 -25
  2. package/dist/alignment/DTWMfccSequenceAlignment.d.ts +1 -1
  3. package/dist/alignment/DTWMfccSequenceAlignment.js +5 -5
  4. package/dist/alignment/DTWSequenceAlignmentWindowed.js +1 -3
  5. package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -1
  6. package/dist/alignment/{TextAlignment.d.ts → SemanticTextAlignment.d.ts} +4 -2
  7. package/dist/alignment/SemanticTextAlignment.js +336 -0
  8. package/dist/alignment/SemanticTextAlignment.js.map +1 -0
  9. package/dist/alignment/SpeechAlignment.d.ts +4 -3
  10. package/dist/alignment/SpeechAlignment.js +130 -39
  11. package/dist/alignment/SpeechAlignment.js.map +1 -1
  12. package/dist/api/API.d.ts +7 -3
  13. package/dist/api/API.js +7 -2
  14. package/dist/api/API.js.map +1 -1
  15. package/dist/api/APIOptions.d.ts +4 -1
  16. package/dist/api/Alignment.d.ts +1 -1
  17. package/dist/api/Alignment.js +13 -5
  18. package/dist/api/Alignment.js.map +1 -1
  19. package/dist/api/LanguageDetectionCommon.d.ts +6 -0
  20. package/dist/api/LanguageDetectionCommon.js +2 -0
  21. package/dist/api/LanguageDetectionCommon.js.map +1 -0
  22. package/dist/api/Recognition.js.map +1 -1
  23. package/dist/api/{LanguageDetection.d.ts → SpeechLanguageDetection.d.ts} +1 -25
  24. package/dist/api/{LanguageDetection.js → SpeechLanguageDetection.js} +1 -68
  25. package/dist/api/SpeechLanguageDetection.js.map +1 -0
  26. package/dist/api/{Translation.js → SpeechTranslation.js} +3 -3
  27. package/dist/api/SpeechTranslation.js.map +1 -0
  28. package/dist/api/Synthesis.d.ts +0 -1
  29. package/dist/api/Synthesis.js +4 -4
  30. package/dist/api/TextLanguageDetection.d.ts +21 -0
  31. package/dist/api/TextLanguageDetection.js +67 -0
  32. package/dist/api/TextLanguageDetection.js.map +1 -0
  33. package/dist/api/TextTranslation.d.ts +25 -0
  34. package/dist/api/TextTranslation.js +101 -0
  35. package/dist/api/TextTranslation.js.map +1 -0
  36. package/dist/api/TimelineTranslationAlignment.d.ts +23 -0
  37. package/dist/api/TimelineTranslationAlignment.js +92 -0
  38. package/dist/api/TimelineTranslationAlignment.js.map +1 -0
  39. package/dist/api/TranscriptAndTranslationAlignment.d.ts +35 -0
  40. package/dist/api/TranscriptAndTranslationAlignment.js +78 -0
  41. package/dist/api/TranscriptAndTranslationAlignment.js.map +1 -0
  42. package/dist/api/TranslationAlignment.d.ts +4 -3
  43. package/dist/api/TranslationAlignment.js +9 -8
  44. package/dist/api/TranslationAlignment.js.map +1 -1
  45. package/dist/api/VoiceActivityDetection.js +16 -1
  46. package/dist/api/VoiceActivityDetection.js.map +1 -1
  47. package/dist/audio/AudioBufferConversion.d.ts +0 -1
  48. package/dist/audio/AudioPlayer.d.ts +0 -1
  49. package/dist/audio/AudioPlayer.js +62 -41
  50. package/dist/audio/AudioPlayer.js.map +1 -1
  51. package/dist/audio/AudioUtilities.d.ts +0 -1
  52. package/dist/cli/CLI.d.ts +28 -7
  53. package/dist/cli/CLI.js +265 -37
  54. package/dist/cli/CLI.js.map +1 -1
  55. package/dist/codecs/FFMpegTranscoder.d.ts +0 -1
  56. package/dist/codecs/FFMpegTranscoder.js +7 -0
  57. package/dist/codecs/FFMpegTranscoder.js.map +1 -1
  58. package/dist/codecs/TIMITCodec.d.ts +0 -1
  59. package/dist/codecs/WaveCodec.d.ts +0 -1
  60. package/dist/dsp/FFT.d.ts +1 -1
  61. package/dist/dsp/FFT.js +6 -0
  62. package/dist/dsp/FFT.js.map +1 -1
  63. package/dist/dsp/KWeightingFilter.js +1 -1
  64. package/dist/dsp/KWeightingFilter.js.map +1 -1
  65. package/dist/dsp/MelSpectogram.d.ts +3 -2
  66. package/dist/dsp/MelSpectogram.js +14 -8
  67. package/dist/dsp/MelSpectogram.js.map +1 -1
  68. package/dist/math/VectorMath.d.ts +9 -9
  69. package/dist/math/VectorMath.js +10 -10
  70. package/dist/math/VectorMath.js.map +1 -1
  71. package/dist/nlp/ChineseSegmentation.js +4 -4
  72. package/dist/nlp/ChineseSegmentation.js.map +1 -1
  73. package/dist/nlp/Segmentation.d.ts +2 -2
  74. package/dist/nlp/Segmentation.js +20 -13
  75. package/dist/nlp/Segmentation.js.map +1 -1
  76. package/dist/recognition/OpenAICloudSTT.d.ts +2 -1
  77. package/dist/recognition/OpenAICloudSTT.js +30 -19
  78. package/dist/recognition/OpenAICloudSTT.js.map +1 -1
  79. package/dist/recognition/SileroSTT.d.ts +0 -1
  80. package/dist/recognition/WhisperCppSTT.d.ts +3 -3
  81. package/dist/recognition/WhisperCppSTT.js +21 -9
  82. package/dist/recognition/WhisperCppSTT.js.map +1 -1
  83. package/dist/recognition/WhisperSTT.d.ts +9 -6
  84. package/dist/recognition/WhisperSTT.js +227 -46
  85. package/dist/recognition/WhisperSTT.js.map +1 -1
  86. package/dist/server/Client.d.ts +3 -4
  87. package/dist/server/Client.js.map +1 -1
  88. package/dist/server/Worker.d.ts +3 -3
  89. package/dist/server/Worker.js +3 -2
  90. package/dist/server/Worker.js.map +1 -1
  91. package/dist/source-separation/MDXNetSourceSeparation.d.ts +0 -1
  92. package/dist/source-separation/MDXNetSourceSeparation.js +1 -1
  93. package/dist/source-separation/MDXNetSourceSeparation.js.map +1 -1
  94. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.d.ts +12 -0
  95. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js +68 -0
  96. package/dist/speech-embeddings/WavToVec2BertFeatureEmbeddings.js.map +1 -0
  97. package/dist/speech-language-detection/SileroLanguageDetection.d.ts +1 -2
  98. package/dist/speech-language-detection/SileroLanguageDetection.js +1 -1
  99. package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -1
  100. package/dist/subtitles/Subtitles.js +2 -2
  101. package/dist/subtitles/Subtitles.js.map +1 -1
  102. package/dist/synthesis/GoogleCloudTTS.d.ts +0 -1
  103. package/dist/synthesis/GoogleTranslateTTS.d.ts +0 -1
  104. package/dist/synthesis/GoogleTranslateTTS.js +6 -21
  105. package/dist/synthesis/GoogleTranslateTTS.js.map +1 -1
  106. package/dist/synthesis/StreamlabsPollyTTS.d.ts +0 -1
  107. package/dist/synthesis/VitsTTS.d.ts +0 -1
  108. package/dist/synthesis/VitsTTS.js +30 -0
  109. package/dist/synthesis/VitsTTS.js.map +1 -1
  110. package/dist/tests/Test.js +0 -31
  111. package/dist/tests/Test.js.map +1 -1
  112. package/dist/text-language-detection/FastTextLanguageDetection.d.ts +1 -1
  113. package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -1
  114. package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +1 -1
  115. package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -1
  116. package/dist/text-translation/DeepLTextTranslation.d.ts +2 -0
  117. package/dist/text-translation/DeepLTextTranslation.js +67 -0
  118. package/dist/text-translation/DeepLTextTranslation.js.map +1 -0
  119. package/dist/text-translation/GoogleTranslateTextTranslation.d.ts +10 -0
  120. package/dist/text-translation/GoogleTranslateTextTranslation.js +554 -0
  121. package/dist/text-translation/GoogleTranslateTextTranslation.js.map +1 -0
  122. package/dist/text-translation/NLLBTextTranslation.d.ts +2 -1
  123. package/dist/text-translation/NLLBTextTranslation.js +249 -19
  124. package/dist/text-translation/NLLBTextTranslation.js.map +1 -1
  125. package/dist/utilities/BinaryArrayConversion.d.ts +0 -1
  126. package/dist/utilities/BrowserRequestHeaders.d.ts +6 -0
  127. package/dist/utilities/BrowserRequestHeaders.js +52 -0
  128. package/dist/utilities/BrowserRequestHeaders.js.map +1 -0
  129. package/dist/utilities/BufferFileReadStream.d.ts +20 -0
  130. package/dist/utilities/BufferFileReadStream.js +81 -0
  131. package/dist/utilities/BufferFileReadStream.js.map +1 -0
  132. package/dist/utilities/DynamicUint8Array.d.ts +9 -0
  133. package/dist/utilities/DynamicUint8Array.js +31 -0
  134. package/dist/utilities/DynamicUint8Array.js.map +1 -0
  135. package/dist/utilities/FileSystem.d.ts +0 -2
  136. package/dist/utilities/Hashing.d.ts +3 -10
  137. package/dist/utilities/Hashing.js +10 -127
  138. package/dist/utilities/Hashing.js.map +1 -1
  139. package/dist/utilities/LEB128.d.ts +15 -5
  140. package/dist/utilities/LEB128.js +199 -119
  141. package/dist/utilities/LEB128.js.map +1 -1
  142. package/dist/utilities/LPVarInt.d.ts +11 -0
  143. package/dist/utilities/LPVarInt.js +187 -0
  144. package/dist/utilities/LPVarInt.js.map +1 -0
  145. package/dist/utilities/Locale.d.ts +1 -1
  146. package/dist/utilities/Locale.js +1 -1
  147. package/dist/utilities/OnnxUtilities.d.ts +1 -2
  148. package/dist/utilities/PVarInt.d.ts +4 -0
  149. package/dist/utilities/PVarInt.js +166 -0
  150. package/dist/utilities/PVarInt.js.map +1 -0
  151. package/dist/utilities/PackageManager.js +48 -25
  152. package/dist/utilities/PackageManager.js.map +1 -1
  153. package/dist/utilities/RandomGenerator.d.ts +3 -17
  154. package/dist/utilities/RandomGenerator.js +12 -81
  155. package/dist/utilities/RandomGenerator.js.map +1 -1
  156. package/dist/utilities/Timeline.d.ts +2 -0
  157. package/dist/utilities/Timeline.js +129 -20
  158. package/dist/utilities/Timeline.js.map +1 -1
  159. package/dist/utilities/Utilities.d.ts +1 -3
  160. package/dist/utilities/Utilities.js +30 -3
  161. package/dist/utilities/Utilities.js.map +1 -1
  162. package/dist/utilities/VarInt.d.ts +4 -0
  163. package/dist/utilities/VarInt.js +166 -0
  164. package/dist/utilities/VarInt.js.map +1 -0
  165. package/dist/utilities/VirtualFileReadStream.d.ts +20 -0
  166. package/dist/utilities/VirtualFileReadStream.js +79 -0
  167. package/dist/utilities/VirtualFileReadStream.js.map +1 -0
  168. package/dist/utilities/WebReader.js +7 -23
  169. package/dist/utilities/WebReader.js.map +1 -1
  170. package/dist/voice-activity-detection/SileroVAD.d.ts +0 -1
  171. package/docs/API.md +105 -3
  172. package/docs/CLI.md +51 -1
  173. package/docs/Engines.md +32 -3
  174. package/docs/Options.md +53 -12
  175. package/docs/Tasklist.md +1 -13
  176. package/package.json +20 -24
  177. package/src/alignment/DTWMfccSequenceAlignment.ts +5 -5
  178. package/src/alignment/DTWSequenceAlignmentWindowed.ts +1 -3
  179. package/src/alignment/SemanticTextAlignment.ts +467 -0
  180. package/src/alignment/SpeechAlignment.ts +214 -56
  181. package/src/api/API.ts +18 -2
  182. package/src/api/APIOptions.ts +14 -1
  183. package/src/api/Alignment.ts +31 -9
  184. package/src/api/LanguageDetectionCommon.ts +7 -0
  185. package/src/api/Recognition.ts +2 -0
  186. package/src/api/{LanguageDetection.ts → SpeechLanguageDetection.ts} +1 -119
  187. package/src/api/{Translation.ts → SpeechTranslation.ts} +2 -2
  188. package/src/api/Synthesis.ts +4 -4
  189. package/src/api/TextLanguageDetection.ts +116 -0
  190. package/src/api/TextTranslation.ts +177 -0
  191. package/src/api/TimelineTranslationAlignment.ts +162 -0
  192. package/src/api/TranscriptAndTranslationAlignment.ts +164 -0
  193. package/src/api/TranslationAlignment.ts +12 -10
  194. package/src/api/VoiceActivityDetection.ts +24 -3
  195. package/src/audio/AudioPlayer.ts +2 -0
  196. package/src/cli/CLI.ts +376 -40
  197. package/src/codecs/FFMpegTranscoder.ts +6 -0
  198. package/src/dsp/FFT.ts +8 -2
  199. package/src/dsp/KWeightingFilter.ts +1 -1
  200. package/src/dsp/MelSpectogram.ts +17 -8
  201. package/src/math/VectorMath.ts +15 -15
  202. package/src/nlp/ChineseSegmentation.ts +6 -4
  203. package/src/nlp/Segmentation.ts +18 -13
  204. package/src/recognition/OpenAICloudSTT.ts +47 -29
  205. package/src/recognition/WhisperCppSTT.ts +26 -11
  206. package/src/recognition/WhisperSTT.ts +364 -49
  207. package/src/server/Client.ts +3 -2
  208. package/src/server/Worker.ts +3 -2
  209. package/src/source-separation/MDXNetSourceSeparation.ts +1 -1
  210. package/src/speech-embeddings/WavToVec2BertFeatureEmbeddings.ts +107 -0
  211. package/src/speech-language-detection/SileroLanguageDetection.ts +2 -1
  212. package/src/subtitles/Subtitles.ts +2 -2
  213. package/src/synthesis/GoogleTranslateTTS.ts +7 -21
  214. package/src/synthesis/VitsTTS.ts +31 -3
  215. package/src/tests/Test.ts +1 -38
  216. package/src/text-language-detection/FastTextLanguageDetection.ts +1 -1
  217. package/src/text-language-detection/TinyLDLanguageDetection.ts +1 -1
  218. package/src/text-translation/DeepLTextTranslation.ts +88 -0
  219. package/src/text-translation/GoogleTranslateTextTranslation.ts +667 -0
  220. package/src/text-translation/NLLBTextTranslation.ts +261 -21
  221. package/src/typings/Fillers.d.ts +25 -2
  222. package/src/utilities/BrowserRequestHeaders.ts +59 -0
  223. package/src/utilities/DynamicUint8Array.ts +39 -0
  224. package/src/utilities/Hashing.ts +14 -167
  225. package/src/utilities/LEB128.ts +273 -148
  226. package/src/utilities/LPVarInt.ts +292 -0
  227. package/src/utilities/Locale.ts +1 -1
  228. package/src/utilities/OnnxUtilities.ts +1 -1
  229. package/src/utilities/PackageManager.ts +51 -30
  230. package/src/utilities/RandomGenerator.ts +12 -113
  231. package/src/utilities/Timeline.ts +162 -23
  232. package/src/utilities/Utilities.ts +40 -3
  233. package/src/utilities/VirtualFileReadStream.ts +109 -0
  234. package/src/utilities/WebReader.ts +9 -23
  235. package/dist/alignment/TextAlignment.js +0 -156
  236. package/dist/alignment/TextAlignment.js.map +0 -1
  237. package/dist/api/LanguageDetection.js.map +0 -1
  238. package/dist/api/Translation.js.map +0 -1
  239. package/src/alignment/TextAlignment.ts +0 -234
  240. /package/dist/api/{Translation.d.ts → SpeechTranslation.d.ts} +0 -0
@@ -10,7 +10,7 @@ export class KWeightingFilter {
10
10
 
11
11
  if (useStandard44100Filters) {
12
12
  // These parameter values are taken from ITU-R BS.1770-2
13
- // and designed only for a 44100 Hz sampling rate:
13
+ // and designed only for a 44100 Hz sample rate:
14
14
  this.highShelfFilter = new BiquadFilter({
15
15
  b0: 1.53512485958697,
16
16
  b1: -2.69169618940638,
@@ -2,7 +2,7 @@ import { RawAudio } from '../audio/AudioUtilities.js'
2
2
  import { Logger } from '../utilities/Logger.js'
3
3
  import * as FFT from './FFT.js'
4
4
 
5
- export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbankCount: number, lowerFrequencyHz: number, upperFrequencyHz: number) {
5
+ export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbankCount: number, lowerFrequencyHz: number, upperFrequencyHz: number, windowType: FFT.WindowType = 'hann') {
6
6
  const logger = new Logger()
7
7
 
8
8
  logger.start('Compute mel filterbank')
@@ -18,15 +18,15 @@ export async function computeMelSpectogram(rawAudio: RawAudio, fftOrder: number,
18
18
 
19
19
  logger.end()
20
20
 
21
- return computeMelSpectogramUsingFilterbanks(rawAudio, fftOrder, windowSize, hopLength, melFilterbanks)
21
+ return computeMelSpectogramUsingFilterbanks(rawAudio, fftOrder, windowSize, hopLength, melFilterbanks, windowType)
22
22
  }
23
23
 
24
- export async function computeMelSpectogramUsingFilterbanks(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbanks: Filterbank[]) {
24
+ export async function computeMelSpectogramUsingFilterbanks(rawAudio: RawAudio, fftOrder: number, windowSize: number, hopLength: number, filterbanks: Filterbank[], windowType: FFT.WindowType = 'hann') {
25
25
  const logger = new Logger()
26
26
 
27
27
  logger.start('Compute short-time FFTs')
28
28
  const audioSamples = rawAudio.audioChannels[0]
29
- const fftFrames = await FFT.stftr(audioSamples, fftOrder, windowSize, hopLength, 'hann')
29
+ const fftFrames = await FFT.stftr(audioSamples, fftOrder, windowSize, hopLength, windowType)
30
30
 
31
31
  logger.start('Convert FFT frames to a mel spectogram')
32
32
  const melSpectogram = fftFramesToMelSpectogram(fftFrames, filterbanks)
@@ -52,17 +52,26 @@ export function powerSpectrumToMelSpectrum(powerSpectrum: Float32Array, filterba
52
52
  const filterbankStartIndex = filterbank.startIndex
53
53
  const filterbankWeights = filterbank.weights
54
54
 
55
- if (filterbankStartIndex == -1) {
55
+ if (filterbankStartIndex === -1) {
56
56
  continue
57
57
  }
58
58
 
59
- let bandValue = 0
59
+ let melBandValue = 0
60
60
 
61
61
  for (let i = 0; i < filterbankWeights.length; i++) {
62
- bandValue += filterbankWeights[i] * powerSpectrum[filterbankStartIndex + i]
62
+ const powerSpectrumIndex = filterbankStartIndex + i
63
+
64
+ if (powerSpectrumIndex >= powerSpectrum.length) {
65
+ break
66
+ }
67
+
68
+ const weight = filterbankWeights[i]
69
+ const powerSpectrumValue = powerSpectrum[powerSpectrumIndex]
70
+
71
+ melBandValue += weight * powerSpectrumValue
63
72
  }
64
73
 
65
- melSpectrum[melBandIndex] = bandValue
74
+ melSpectrum[melBandIndex] = melBandValue
66
75
  }
67
76
 
68
77
  return melSpectrum
@@ -177,7 +177,7 @@ export function scaleToSumTo1(vector: number[]) {
177
177
  return scaledVector
178
178
  }
179
179
 
180
- export function normalizeVector(vector: number[], kind: 'population' | 'sample' = 'population') {
180
+ export function normalizeVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population') {
181
181
  if (vector.length == 0) {
182
182
  throw new Error('Vector is empty')
183
183
  }
@@ -329,15 +329,15 @@ export function varianceOfVectors(vectors: number[][], kind: 'population' | 'sam
329
329
  return result
330
330
  }
331
331
 
332
- export function meanOfVector(vector: number[]) {
332
+ export function meanOfVector(vector: ArrayLike<number>) {
333
333
  if (vector.length == 0) {
334
- throw new Error('Vector is empty')
334
+ return 0
335
335
  }
336
336
 
337
337
  return sumVector(vector) / vector.length
338
338
  }
339
339
 
340
- export function medianOfVector(vector: number[]) {
340
+ export function medianOfVector(vector: ArrayLike<number>) {
341
341
  if (vector.length == 0) {
342
342
  throw new Error('Vector is empty')
343
343
  }
@@ -345,13 +345,13 @@ export function medianOfVector(vector: number[]) {
345
345
  return vector[Math.floor(vector.length / 2)]
346
346
  }
347
347
 
348
- export function stdDeviationOfVector(vector: number[], kind: 'population' | 'sample' = 'population', mean?: number) {
348
+ export function stdDeviationOfVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population', mean?: number) {
349
349
  return Math.sqrt(varianceOfVector(vector, kind, mean))
350
350
  }
351
351
 
352
- export function varianceOfVector(vector: number[], kind: 'population' | 'sample' = 'population', mean?: number) {
352
+ export function varianceOfVector(vector: ArrayLike<number>, kind: 'population' | 'sample' = 'population', mean?: number) {
353
353
  if (vector.length == 0) {
354
- throw new Error('Vector is empty')
354
+ return 0
355
355
  }
356
356
 
357
357
  const sampleSizeMetric = kind == 'population' || vector.length == 1 ? vector.length : vector.length - 1
@@ -362,8 +362,8 @@ export function varianceOfVector(vector: number[], kind: 'population' | 'sample'
362
362
 
363
363
  let result = 0.0
364
364
 
365
- for (const value of vector) {
366
- result += (value - mean) ** 2
365
+ for (let i = 0; i < vector.length; i++) {
366
+ result += (vector[i] - mean) ** 2
367
367
  }
368
368
 
369
369
  return result / sampleSizeMetric
@@ -456,11 +456,11 @@ export function meanSquaredError(actual: ArrayLike<number>, expected: ArrayLike<
456
456
  return sum / featureCount
457
457
  }
458
458
 
459
- export function euclidianDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
460
- return Math.sqrt(squaredEuclidianDistance(vector1, vector2))
459
+ export function euclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
460
+ return Math.sqrt(squaredEuclideanDistance(vector1, vector2))
461
461
  }
462
462
 
463
- export function squaredEuclidianDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
463
+ export function squaredEuclideanDistance(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
464
464
  if (vector1.length !== vector2.length) {
465
465
  throw new Error('Vectors are not the same length')
466
466
  }
@@ -480,11 +480,11 @@ export function squaredEuclidianDistance(vector1: ArrayLike<number>, vector2: Ar
480
480
  return sum
481
481
  }
482
482
 
483
- export function euclidianDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
484
- return Math.sqrt(squaredEuclidianDistance13Dim(vector1, vector2))
483
+ export function euclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
484
+ return Math.sqrt(squaredEuclideanDistance13Dim(vector1, vector2))
485
485
  }
486
486
 
487
- export function squaredEuclidianDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
487
+ export function squaredEuclideanDistance13Dim(vector1: ArrayLike<number>, vector2: ArrayLike<number>) {
488
488
  // Assumes the input has 13 dimensions (optimized for 13-dimensional MFCC vectors)
489
489
 
490
490
  const result =
@@ -1,5 +1,5 @@
1
1
  export async function splitChineseTextToWords_Jieba(text: string, fineGrained = false, useHMM = true) {
2
- const jieba = await getWasmInstance()
2
+ const jieba = await getJiebaWasmInstance()
3
3
 
4
4
  if (!fineGrained) {
5
5
  return jieba.cut(text, useHMM)
@@ -58,10 +58,12 @@ export async function splitChineseTextToWords_Jieba(text: string, fineGrained =
58
58
  }
59
59
 
60
60
  let JiebaWasmInstance: typeof import('jieba-wasm')
61
- async function getWasmInstance() {
61
+
62
+ async function getJiebaWasmInstance() {
62
63
  if (!JiebaWasmInstance) {
63
- const { default: JibeaWasm } = await import('jieba-wasm')
64
- JiebaWasmInstance = JibeaWasm
64
+ const { default: JiebaWasm } = await import('jieba-wasm')
65
+
66
+ JiebaWasmInstance = JiebaWasm
65
67
  }
66
68
 
67
69
  return JiebaWasmInstance
@@ -11,8 +11,7 @@ const log = logToStderr
11
11
  export const wordCharacterPattern = /[\p{Letter}\p{Number}]/u
12
12
  export const punctuationPattern = /[\p{Punctuation}]/u
13
13
 
14
- export const phraseSeparators = [',', ';', ':']
15
- export const sentenceSeparators = ['.', '?', '!']
14
+ export const phraseSeparators = [',', ';', ':', ',', '、']
16
15
  export const symbolWords = ['$', '€', '¢', '£', '¥', '©', '®', '™', '%', '&', '#', '~', '@', '+', '±', '÷', '/', '*', '=', '¼', '½', '¾']
17
16
 
18
17
  export function isWordOrSymbolWord(str: string) {
@@ -226,29 +225,35 @@ export async function splitToWords(text: string, langCode: string): Promise<stri
226
225
  }
227
226
  }
228
227
 
229
- export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType, whitespace: WhitespaceProcessing) {
228
+ export function splitToParagraphs(text: string, paragraphBreaks: ParagraphBreakType, whitespaceProcessingMethod: WhitespaceProcessing) {
230
229
  let paragraphs: string[] = []
231
230
 
232
- if (paragraphBreaks == 'single') {
231
+ if (paragraphBreaks === 'single') {
233
232
  paragraphs = text.split(/(\r?\n)+/g)
234
- } else if (paragraphBreaks == 'double') {
233
+ } else if (paragraphBreaks === 'double') {
235
234
  paragraphs = text.split(/(\r?\n)(\r?\n)+/g)
236
235
  } else {
237
- throw new Error(`Invalid paragraph break type: ${paragraphBreaks}`)
236
+ throw new Error(`Invalid paragraph break type: '${paragraphBreaks}'`)
238
237
  }
239
238
 
240
- if (whitespace == 'removeLineBreaks') {
241
- paragraphs = paragraphs.map(p => p.replaceAll(/(\r?\n)+/g, ' '))
242
- } else if (whitespace == 'collapse') {
243
- paragraphs = paragraphs.map(p => p.replaceAll(/\s+/g, ' '))
244
- }
245
-
246
- paragraphs = paragraphs.map(p => p.trim())
239
+ paragraphs = paragraphs.map(p => applyWhitespaceProcessing(p.trim(), whitespaceProcessingMethod))
247
240
  paragraphs = paragraphs.filter(p => p.length > 0)
248
241
 
249
242
  return paragraphs
250
243
  }
251
244
 
245
+ export function applyWhitespaceProcessing(text: string, whitespaceProcessingMethod: WhitespaceProcessing) {
246
+ if (whitespaceProcessingMethod === 'removeLineBreaks') {
247
+ return text.replaceAll(/(\r?\n)+/g, ' ')
248
+ } else if (whitespaceProcessingMethod === 'collapse') {
249
+ return text.replaceAll(/\s+/g, ' ')
250
+ } else if (whitespaceProcessingMethod === 'preserve') {
251
+ return text
252
+ } else {
253
+ throw new Error(`Invalid whitespace processing method: '${whitespaceProcessingMethod}'`)
254
+ }
255
+ }
256
+
252
257
  export function splitToLines(text: string) {
253
258
  return text.split(/\r?\n/g)
254
259
  }
@@ -1,8 +1,10 @@
1
- import { RawAudio } from '../audio/AudioUtilities.js';
2
1
  import * as FFMpegTranscoder from '../codecs/FFMpegTranscoder.js'
3
- import { Logger } from '../utilities/Logger.js';
4
- import { extendDeep } from '../utilities/ObjectUtilities.js';
5
- import { Timeline, TimelineEntry } from '../utilities/Timeline.js';
2
+ import { RawAudio } from '../audio/AudioUtilities.js'
3
+ import { createVirtualFileReadStreamForBuffer } from '../utilities/VirtualFileReadStream.js'
4
+ import { Logger } from '../utilities/Logger.js'
5
+ import { extendDeep } from '../utilities/ObjectUtilities.js'
6
+ import { Timeline, TimelineEntry } from '../utilities/Timeline.js'
7
+ import { alignSegments } from '../api/Alignment.js'
6
8
 
7
9
  export async function recognize(rawAudio: RawAudio, languageCode: string, options: OpenAICloudSTTOptions, task: Task = 'transcribe') {
8
10
  const logger = new Logger()
@@ -11,41 +13,56 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
11
13
 
12
14
  options = extendDeep(defaultOpenAICloudSTTOptions, options)
13
15
 
16
+ if (options.requestWordTimestamps === undefined) {
17
+ options.requestWordTimestamps = options.baseURL === undefined
18
+ }
19
+
20
+ if (options.model === undefined) {
21
+ if (options.baseURL === undefined) {
22
+ options.model = 'whisper-1'
23
+ } else {
24
+ throw new Error(`A custom provider for the OpenAI Cloud API requires specifying a model name`)
25
+ }
26
+ }
27
+
14
28
  const { default: OpenAI } = await import('openai')
15
29
  const openai = new OpenAI(options)
16
30
 
17
31
  logger.start('Encode audio to send')
18
32
  const ffmpegOptions = FFMpegTranscoder.getDefaultFFMpegOptionsForSpeech('mp3')
19
33
  const encodedAudio = await FFMpegTranscoder.encodeFromChannels(rawAudio, ffmpegOptions)
20
- const audioAsWaveBlob = new FileLikeBlob([encodedAudio], 'audio', Date.now(), { type: 'audio/mpeg' })
34
+ const virtualFileStream = createVirtualFileReadStreamForBuffer(encodedAudio, 'audio.mp3')
21
35
 
22
- logger.start('Request recognition from OpenAI Cloud API')
36
+ logger.start(options.baseURL ? `Send request to ${options.baseURL}` : 'Send request to OpenAI Cloud API')
23
37
 
24
38
  let response: VerboseResponse
25
39
 
26
- if (task =='transcribe') {
40
+ if (task == 'transcribe') {
41
+ const timestamp_granularities: ('word' | 'segment')[] | undefined =
42
+ options.requestWordTimestamps ? ['word', 'segment'] : undefined
43
+
27
44
  response = await openai.audio.transcriptions.create({
28
- file: audioAsWaveBlob,
29
- model: options.model!,
45
+ file: virtualFileStream,
46
+ model: options.model,
30
47
  language: languageCode,
31
48
  prompt: options.prompt,
32
49
  response_format: 'verbose_json',
33
50
  temperature: options.temperature,
34
- timestamp_granularities: ['word', 'segment']
35
- }) as VerboseResponse
51
+ timestamp_granularities,
52
+ }) as any as VerboseResponse
36
53
  } else if (task == 'translate') {
37
54
  response = await openai.audio.translations.create({
38
- file: audioAsWaveBlob,
39
- model: options.model!,
55
+ file: virtualFileStream,
56
+ model: options.model,
40
57
  prompt: options.prompt,
41
58
  response_format: 'verbose_json',
42
59
  temperature: options.temperature,
43
- }) as VerboseResponse
60
+ }) as any as VerboseResponse
44
61
  } else {
45
62
  throw new Error(`Invalid task`)
46
63
  }
47
64
 
48
- const transcript = response.text
65
+ const transcript = response.text.trim()
49
66
 
50
67
  let timeline: Timeline
51
68
 
@@ -57,12 +74,20 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
57
74
  endTime: entry.end
58
75
  }))
59
76
  } else {
60
- timeline = response.segments.map<TimelineEntry>(entry => ({
77
+ const segmentTimeline = response.segments.map<TimelineEntry>(entry => ({
61
78
  type: 'segment',
62
79
  text: entry.text,
63
80
  startTime: entry.start,
64
81
  endTime: entry.end
65
82
  }))
83
+
84
+ if (task === 'transcribe') {
85
+ logger.start('Align segments')
86
+
87
+ timeline = await alignSegments(rawAudio, segmentTimeline, { language: languageCode })
88
+ } else {
89
+ timeline = segmentTimeline
90
+ }
66
91
  }
67
92
 
68
93
  logger.end()
@@ -70,17 +95,6 @@ export async function recognize(rawAudio: RawAudio, languageCode: string, option
70
95
  return { transcript, timeline }
71
96
  }
72
97
 
73
- class FileLikeBlob extends Blob {
74
- constructor(
75
- public readonly parts: BlobPart[],
76
- public readonly name: string,
77
- public readonly lastModified: number,
78
- options: BlobPropertyBag,
79
- ) {
80
- super(parts, options)
81
- }
82
- }
83
-
84
98
  interface VerboseResponse {
85
99
  task: string
86
100
  language: string
@@ -115,7 +129,7 @@ interface VerboseResponse {
115
129
  type Task = 'transcribe' | 'translate'
116
130
 
117
131
  export interface OpenAICloudSTTOptions {
118
- model?: 'whisper-1'
132
+ model?: 'whisper-1' | string
119
133
 
120
134
  apiKey?: string
121
135
  organization?: string
@@ -126,6 +140,8 @@ export interface OpenAICloudSTTOptions {
126
140
 
127
141
  timeout?: number
128
142
  maxRetries?: number
143
+
144
+ requestWordTimestamps?: boolean
129
145
  }
130
146
 
131
147
  export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
@@ -133,10 +149,12 @@ export const defaultOpenAICloudSTTOptions: OpenAICloudSTTOptions = {
133
149
  organization: undefined,
134
150
  baseURL: undefined,
135
151
 
136
- model: 'whisper-1',
152
+ model: undefined,
137
153
  temperature: 0,
138
154
  prompt: undefined,
139
155
 
140
156
  timeout: undefined,
141
157
  maxRetries: 10,
158
+
159
+ requestWordTimestamps: undefined,
142
160
  }
@@ -13,7 +13,7 @@ import { splitToLines } from '../nlp/Segmentation.js'
13
13
  import { extendDeep } from '../utilities/ObjectUtilities.js'
14
14
  import { formatLanguageCodeWithName, getShortLanguageCode } from '../utilities/Locale.js'
15
15
  import { loadPackage } from '../utilities/PackageManager.js'
16
- import { detectSpeechLanguageByParts } from '../api/LanguageDetection.js'
16
+ import { detectSpeechLanguageByParts } from '../api/SpeechLanguageDetection.js'
17
17
 
18
18
  export async function recognize(
19
19
  sourceRawAudio: RawAudio,
@@ -54,7 +54,7 @@ export async function recognize(
54
54
  }
55
55
  } else {
56
56
  if (options.enableGPU) {
57
- buildKind = 'cublas-11.8.0'
57
+ buildKind = 'cublas-12.4.0'
58
58
  } else {
59
59
  buildKind = 'cpu'
60
60
  }
@@ -201,7 +201,7 @@ export async function recognize(
201
201
 
202
202
  export async function detectLanguage(sourceRawAudio: RawAudio, modelName: WhisperModelName, modelPath: string) {
203
203
  if (sourceRawAudio.sampleRate != 16000) {
204
- throw new Error('Source audio must have a sampling rate of 16000')
204
+ throw new Error('Source audio must have a sample rate of 16000')
205
205
  }
206
206
 
207
207
  async function detectLanguageForPart(partAudio: RawAudio) {
@@ -240,6 +240,8 @@ async function parseResultObject(resultObject: WhisperCppVerboseResult, modelNam
240
240
 
241
241
  let currentCorrectionTimeOffset = 0
242
242
 
243
+ let lastTokenEndOffset = 0
244
+
243
245
  for (let segmentIndex = 0; segmentIndex < resultObject.transcription.length; segmentIndex++) {
244
246
  const segmentObject = resultObject.transcription[segmentIndex]
245
247
 
@@ -248,6 +250,17 @@ async function parseResultObject(resultObject: WhisperCppVerboseResult, modelNam
248
250
  for (let tokenIndex = 0; tokenIndex < tokens.length; tokenIndex++) {
249
251
  const tokenObject = tokens[tokenIndex]
250
252
 
253
+ // Workaround whisper.cpp issue with missing offsets by falling back to last known end offset
254
+ // when they are not included
255
+ if (!tokenObject.offsets) {
256
+ tokenObject.offsets = {
257
+ from: lastTokenEndOffset,
258
+ to: lastTokenEndOffset,
259
+ }
260
+ } else {
261
+ lastTokenEndOffset = tokenObject.offsets.to
262
+ }
263
+
251
264
  if (tokenIndex === 0 && tokenObject.text === '[_BEG_]' && tokenObject.offsets.from === 0) {
252
265
  currentCorrectionTimeOffset = segmentObject.offsets.from / 1000
253
266
  }
@@ -370,7 +383,7 @@ export async function loadModelPackage(modelId: WhisperCppModelId | undefined, l
370
383
  return { modelName, modelPath }
371
384
  }
372
385
 
373
- export type WhisperCppBuild = 'cpu' | 'cublas-11.8.0' | 'cublas-12.4.0' | 'custom'
386
+ export type WhisperCppBuild = 'cpu' | 'cublas-12.4.0' | 'custom'
374
387
 
375
388
  export async function loadExecutablePackage(buildKind: WhisperCppBuild) {
376
389
  if (buildKind === 'custom') {
@@ -384,20 +397,20 @@ export async function loadExecutablePackage(buildKind: WhisperCppBuild) {
384
397
 
385
398
  if (buildKind.startsWith('cublas-')) {
386
399
  if (platform === 'win32' && arch === 'x64') {
387
- packageName = `whisper.cpp-binaries-windows-x64-${buildKind}-latest-patched`
400
+ packageName = `whisper.cpp-binaries-windows-x64-${buildKind}-latest`
388
401
  } else {
389
- throw new Error(`GPU builds (NVIDIA CUDA only) are currently only available as packages for Windows x64. Please specify a custom path to the binary in the 'executablePath' option.`)
402
+ throw new Error(`whisper.cpp GPU builds (NVIDIA CUDA only) are currently only available as packages for Windows x64. Please specify a custom path to a whisper.cpp 'main' binary in the 'executablePath' option.`)
390
403
  }
391
404
  } else if (buildKind === 'cpu') {
392
405
  if (platform === 'win32' && arch === 'x64') {
393
- packageName = `whisper.cpp-binaries-windows-x64-cpu-latest-patched`
406
+ packageName = `whisper.cpp-binaries-windows-x64-cpu-latest`
394
407
  } else if (platform === 'linux' && arch === 'x64') {
395
- packageName = `whisper.cpp-binaries-linux-x64-cpu-latest-patched`
408
+ packageName = `whisper.cpp-binaries-linux-x64-cpu-latest`
396
409
  } else {
397
- throw new Error(`Couldn't find a matching whisper.cpp binary package. Please specify a custom path to the binary in the 'executablePath' option.`)
410
+ throw new Error(`Couldn't find a matching whisper.cpp binary package. Please specify a custom path to a whisper.cpp 'main' binary in the 'executablePath' option.`)
398
411
  }
399
412
  } else {
400
- throw new Error(`Unknown build kind '${buildKind}'`)
413
+ throw new Error(`Unsupported build kind '${buildKind}'`)
401
414
  }
402
415
 
403
416
  const packagePath = await loadPackage(packageName)
@@ -553,4 +566,6 @@ export type WhisperCppModelId =
553
566
  'large-v2' |
554
567
  'large-v2-q5_0' |
555
568
  'large-v3' |
556
- 'large-v3-q5_0'
569
+ 'large-v3-q5_0' |
570
+ `large-v3-turbo` |
571
+ `large-v3-turbo-q5_0`