echogarden 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -1
- package/data/lexicons/heteronyms.json +674 -0
- package/data/schemas/options.json +1057 -0
- package/data/tables/lcid-table.json +7427 -0
- package/dist/alignment/DTWMfccSequenceAlignment.d.ts +2 -0
- package/dist/alignment/DTWMfccSequenceAlignment.js +26 -0
- package/dist/alignment/DTWMfccSequenceAlignment.js.map +1 -0
- package/dist/alignment/DTWSequenceAlignment.d.ts +5 -0
- package/dist/alignment/DTWSequenceAlignment.js +99 -0
- package/dist/alignment/DTWSequenceAlignment.js.map +1 -0
- package/dist/alignment/DTWSequenceAlignmentWindowed.d.ts +5 -0
- package/dist/alignment/DTWSequenceAlignmentWindowed.js +161 -0
- package/dist/alignment/DTWSequenceAlignmentWindowed.js.map +1 -0
- package/dist/alignment/LevenshteinSequenceAlignment.d.ts +5 -0
- package/dist/alignment/LevenshteinSequenceAlignment.js +102 -0
- package/dist/alignment/LevenshteinSequenceAlignment.js.map +1 -0
- package/dist/alignment/SpeechAlignment.d.ts +27 -0
- package/dist/alignment/SpeechAlignment.js +266 -0
- package/dist/alignment/SpeechAlignment.js.map +1 -0
- package/dist/api/API.d.ts +8 -0
- package/dist/api/API.js +10 -0
- package/dist/api/API.js.map +1 -0
- package/dist/api/APIOptions.d.ts +12 -0
- package/dist/api/APIOptions.js +2 -0
- package/dist/api/APIOptions.js.map +1 -0
- package/dist/api/Alignment.d.ts +32 -0
- package/dist/api/Alignment.js +134 -0
- package/dist/api/Alignment.js.map +1 -0
- package/dist/api/Denoising.d.ts +14 -0
- package/dist/api/Denoising.js +66 -0
- package/dist/api/Denoising.js.map +1 -0
- package/dist/api/Globals.d.ts +1 -0
- package/dist/api/Globals.js +2 -0
- package/dist/api/Globals.js.map +1 -0
- package/dist/api/LanguageDetection.d.ts +49 -0
- package/dist/api/LanguageDetection.js +118 -0
- package/dist/api/LanguageDetection.js.map +1 -0
- package/dist/api/Recognition.d.ts +49 -0
- package/dist/api/Recognition.js +162 -0
- package/dist/api/Recognition.js.map +1 -0
- package/dist/api/Synthesis.d.ts +125 -0
- package/dist/api/Synthesis.js +892 -0
- package/dist/api/Synthesis.js.map +1 -0
- package/dist/api/Translation.d.ts +25 -0
- package/dist/api/Translation.js +69 -0
- package/dist/api/Translation.js.map +1 -0
- package/dist/api/Vad.d.ts +25 -0
- package/dist/api/Vad.js +88 -0
- package/dist/api/Vad.js.map +1 -0
- package/dist/audio/AudioBufferConversion.d.ts +15 -0
- package/dist/audio/AudioBufferConversion.js +228 -0
- package/dist/audio/AudioBufferConversion.js.map +1 -0
- package/dist/audio/AudioPlayer.d.ts +9 -0
- package/dist/audio/AudioPlayer.js +232 -0
- package/dist/audio/AudioPlayer.js.map +1 -0
- package/dist/audio/AudioRecorder.d.ts +3 -0
- package/dist/audio/AudioRecorder.js +68 -0
- package/dist/audio/AudioRecorder.js.map +1 -0
- package/dist/audio/AudioUtilities.d.ts +48 -0
- package/dist/audio/AudioUtilities.js +209 -0
- package/dist/audio/AudioUtilities.js.map +1 -0
- package/dist/audio/SoxPath.d.ts +1 -0
- package/dist/audio/SoxPath.js +19 -0
- package/dist/audio/SoxPath.js.map +1 -0
- package/dist/cli/CLI.d.ts +8 -0
- package/dist/cli/CLI.js +860 -0
- package/dist/cli/CLI.js.map +1 -0
- package/dist/cli/CLIConfigFile.d.ts +3 -0
- package/dist/cli/CLIConfigFile.js +62 -0
- package/dist/cli/CLIConfigFile.js.map +1 -0
- package/dist/cli/CLILauncher.d.ts +2 -0
- package/dist/cli/CLILauncher.js +22 -0
- package/dist/cli/CLILauncher.js.map +1 -0
- package/dist/cli/CLIOptionsSchema.d.ts +5 -0
- package/dist/cli/CLIOptionsSchema.js +36 -0
- package/dist/cli/CLIOptionsSchema.js.map +1 -0
- package/dist/cli/CLIParser.d.ts +6 -0
- package/dist/cli/CLIParser.js +34 -0
- package/dist/cli/CLIParser.js.map +1 -0
- package/dist/cli/CLIStarter.d.ts +1 -0
- package/dist/cli/CLIStarter.js +3 -0
- package/dist/cli/CLIStarter.js.map +1 -0
- package/dist/codecs/FFMpegTranscoder.d.ts +20 -0
- package/dist/codecs/FFMpegTranscoder.js +169 -0
- package/dist/codecs/FFMpegTranscoder.js.map +1 -0
- package/dist/codecs/TIMITCodec.d.ts +9 -0
- package/dist/codecs/TIMITCodec.js +14 -0
- package/dist/codecs/TIMITCodec.js.map +1 -0
- package/dist/codecs/WaveCodec.d.ts +19 -0
- package/dist/codecs/WaveCodec.js +208 -0
- package/dist/codecs/WaveCodec.js.map +1 -0
- package/dist/denoising/RNNoise.d.ts +6 -0
- package/dist/denoising/RNNoise.js +68 -0
- package/dist/denoising/RNNoise.js.map +1 -0
- package/dist/dsp/BiquadFilter.d.ts +26 -0
- package/dist/dsp/BiquadFilter.js +399 -0
- package/dist/dsp/BiquadFilter.js.map +1 -0
- package/dist/dsp/FFT.d.ts +9 -0
- package/dist/dsp/FFT.js +135 -0
- package/dist/dsp/FFT.js.map +1 -0
- package/dist/dsp/MFCC.d.ts +25 -0
- package/dist/dsp/MFCC.js +162 -0
- package/dist/dsp/MFCC.js.map +1 -0
- package/dist/dsp/MelSpectogram.d.ts +19 -0
- package/dist/dsp/MelSpectogram.js +102 -0
- package/dist/dsp/MelSpectogram.js.map +1 -0
- package/dist/dsp/Rubberband.d.ts +52 -0
- package/dist/dsp/Rubberband.js +186 -0
- package/dist/dsp/Rubberband.js.map +1 -0
- package/dist/dsp/Sonic.d.ts +2 -0
- package/dist/dsp/Sonic.js +39 -0
- package/dist/dsp/Sonic.js.map +1 -0
- package/dist/dsp/SpeexResampler.d.ts +3 -0
- package/dist/dsp/SpeexResampler.js +53 -0
- package/dist/dsp/SpeexResampler.js.map +1 -0
- package/dist/math/VectorMath.d.ts +70 -0
- package/dist/math/VectorMath.js +564 -0
- package/dist/math/VectorMath.js.map +1 -0
- package/dist/nlp/ChineseSegmentation.d.ts +1 -0
- package/dist/nlp/ChineseSegmentation.js +53 -0
- package/dist/nlp/ChineseSegmentation.js.map +1 -0
- package/dist/nlp/CompromiseNLP.d.ts +15 -0
- package/dist/nlp/CompromiseNLP.js +66 -0
- package/dist/nlp/CompromiseNLP.js.map +1 -0
- package/dist/nlp/EspeakPhonemizer.d.ts +4 -0
- package/dist/nlp/EspeakPhonemizer.js +133 -0
- package/dist/nlp/EspeakPhonemizer.js.map +1 -0
- package/dist/nlp/IPA.d.ts +19 -0
- package/dist/nlp/IPA.js +113 -0
- package/dist/nlp/IPA.js.map +1 -0
- package/dist/nlp/JapaneseSegmentation.d.ts +1 -0
- package/dist/nlp/JapaneseSegmentation.js +40 -0
- package/dist/nlp/JapaneseSegmentation.js.map +1 -0
- package/dist/nlp/Lexicon.d.ts +17 -0
- package/dist/nlp/Lexicon.js +6 -0
- package/dist/nlp/Lexicon.js.map +1 -0
- package/dist/nlp/PhoneConversion.d.ts +6 -0
- package/dist/nlp/PhoneConversion.js +467 -0
- package/dist/nlp/PhoneConversion.js.map +1 -0
- package/dist/nlp/Segmentation.d.ts +41 -0
- package/dist/nlp/Segmentation.js +158 -0
- package/dist/nlp/Segmentation.js.map +1 -0
- package/dist/nlp/TextNormalizer.d.ts +4 -0
- package/dist/nlp/TextNormalizer.js +69 -0
- package/dist/nlp/TextNormalizer.js.map +1 -0
- package/dist/recognition/AmazonTranscribeSTT.d.ts +6 -0
- package/dist/recognition/AmazonTranscribeSTT.js +79 -0
- package/dist/recognition/AmazonTranscribeSTT.js.map +1 -0
- package/dist/recognition/AzureCognitiveServicesSTT.d.ts +7 -0
- package/dist/recognition/AzureCognitiveServicesSTT.js +51 -0
- package/dist/recognition/AzureCognitiveServicesSTT.js.map +1 -0
- package/dist/recognition/GoogleCloudSTT.d.ts +7 -0
- package/dist/recognition/GoogleCloudSTT.js +66 -0
- package/dist/recognition/GoogleCloudSTT.js.map +1 -0
- package/dist/recognition/SileroSTT.d.ts +9 -0
- package/dist/recognition/SileroSTT.js +125 -0
- package/dist/recognition/SileroSTT.js.map +1 -0
- package/dist/recognition/VoskSTT.d.ts +10 -0
- package/dist/recognition/VoskSTT.js +78 -0
- package/dist/recognition/VoskSTT.js.map +1 -0
- package/dist/recognition/WhisperSTT.d.ts +69 -0
- package/dist/recognition/WhisperSTT.js +977 -0
- package/dist/recognition/WhisperSTT.js.map +1 -0
- package/dist/server/Server.d.ts +1 -0
- package/dist/server/Server.js +18 -0
- package/dist/server/Server.js.map +1 -0
- package/dist/speech-language-detection/SileroLanguageDetection.d.ts +9 -0
- package/dist/speech-language-detection/SileroLanguageDetection.js +47 -0
- package/dist/speech-language-detection/SileroLanguageDetection.js.map +1 -0
- package/dist/subtitles/Subtitles.d.ts +25 -0
- package/dist/subtitles/Subtitles.js +286 -0
- package/dist/subtitles/Subtitles.js.map +1 -0
- package/dist/synthesis/AwsPollyTTS.d.ts +7 -0
- package/dist/synthesis/AwsPollyTTS.js +51 -0
- package/dist/synthesis/AwsPollyTTS.js.map +1 -0
- package/dist/synthesis/AzureCognitiveServicesTTS.d.ts +9 -0
- package/dist/synthesis/AzureCognitiveServicesTTS.js +103 -0
- package/dist/synthesis/AzureCognitiveServicesTTS.js.map +1 -0
- package/dist/synthesis/CoquiServerTTS.d.ts +6 -0
- package/dist/synthesis/CoquiServerTTS.js +22 -0
- package/dist/synthesis/CoquiServerTTS.js.map +1 -0
- package/dist/synthesis/ElevenLabsTTS.d.ts +8 -0
- package/dist/synthesis/ElevenLabsTTS.js +48 -0
- package/dist/synthesis/ElevenLabsTTS.js.map +1 -0
- package/dist/synthesis/EspeakTTS.d.ts +46 -0
- package/dist/synthesis/EspeakTTS.js +353 -0
- package/dist/synthesis/EspeakTTS.js.map +1 -0
- package/dist/synthesis/FliteTTS.d.ts +17 -0
- package/dist/synthesis/FliteTTS.js +326 -0
- package/dist/synthesis/FliteTTS.js.map +1 -0
- package/dist/synthesis/GoogleCloudTTS.d.ts +16 -0
- package/dist/synthesis/GoogleCloudTTS.js +72 -0
- package/dist/synthesis/GoogleCloudTTS.js.map +1 -0
- package/dist/synthesis/GoogleTranslateTTS.d.ts +13 -0
- package/dist/synthesis/GoogleTranslateTTS.js +177 -0
- package/dist/synthesis/GoogleTranslateTTS.js.map +1 -0
- package/dist/synthesis/MicrosoftEdgeTTS.d.ts +10 -0
- package/dist/synthesis/MicrosoftEdgeTTS.js +216 -0
- package/dist/synthesis/MicrosoftEdgeTTS.js.map +1 -0
- package/dist/synthesis/SamTTS.d.ts +4 -0
- package/dist/synthesis/SamTTS.js +21 -0
- package/dist/synthesis/SamTTS.js.map +1 -0
- package/dist/synthesis/SapiTTS.d.ts +9 -0
- package/dist/synthesis/SapiTTS.js +211 -0
- package/dist/synthesis/SapiTTS.js.map +1 -0
- package/dist/synthesis/StreamlabsPollyTTS.d.ts +12 -0
- package/dist/synthesis/StreamlabsPollyTTS.js +88 -0
- package/dist/synthesis/StreamlabsPollyTTS.js.map +1 -0
- package/dist/synthesis/SvoxPicoTTS.d.ts +11 -0
- package/dist/synthesis/SvoxPicoTTS.js +236 -0
- package/dist/synthesis/SvoxPicoTTS.js.map +1 -0
- package/dist/synthesis/VitsTTS.d.ts +27 -0
- package/dist/synthesis/VitsTTS.js +359 -0
- package/dist/synthesis/VitsTTS.js.map +1 -0
- package/dist/tests/Test.d.ts +1 -0
- package/dist/tests/Test.js +10 -0
- package/dist/tests/Test.js.map +1 -0
- package/dist/text-language-detection/FastTextLanguageDetection.d.ts +2 -0
- package/dist/text-language-detection/FastTextLanguageDetection.js +38 -0
- package/dist/text-language-detection/FastTextLanguageDetection.js.map +1 -0
- package/dist/text-language-detection/TinyLDLanguageDetection.d.ts +2 -0
- package/dist/text-language-detection/TinyLDLanguageDetection.js +12 -0
- package/dist/text-language-detection/TinyLDLanguageDetection.js.map +1 -0
- package/dist/utilities/BinaryArrayConversion.d.ts +15 -0
- package/dist/utilities/BinaryArrayConversion.js +115 -0
- package/dist/utilities/BinaryArrayConversion.js.map +1 -0
- package/dist/utilities/Compression.d.ts +4 -0
- package/dist/utilities/Compression.js +67 -0
- package/dist/utilities/Compression.js.map +1 -0
- package/dist/utilities/FileDownloader.d.ts +3 -0
- package/dist/utilities/FileDownloader.js +147 -0
- package/dist/utilities/FileDownloader.js.map +1 -0
- package/dist/utilities/FileSystem.d.ts +30 -0
- package/dist/utilities/FileSystem.js +141 -0
- package/dist/utilities/FileSystem.js.map +1 -0
- package/dist/utilities/Hashing.d.ts +10 -0
- package/dist/utilities/Hashing.js +169 -0
- package/dist/utilities/Hashing.js.map +1 -0
- package/dist/utilities/Locale.d.ts +9 -0
- package/dist/utilities/Locale.js +65 -0
- package/dist/utilities/Locale.js.map +1 -0
- package/dist/utilities/Logger.d.ts +11 -0
- package/dist/utilities/Logger.js +49 -0
- package/dist/utilities/Logger.js.map +1 -0
- package/dist/utilities/NdArrayUtilities.d.ts +3 -0
- package/dist/utilities/NdArrayUtilities.js +22 -0
- package/dist/utilities/NdArrayUtilities.js.map +1 -0
- package/dist/utilities/ObjectUtilities.d.ts +4 -0
- package/dist/utilities/ObjectUtilities.js +132 -0
- package/dist/utilities/ObjectUtilities.js.map +1 -0
- package/dist/utilities/OpenPromise.d.ts +6 -0
- package/dist/utilities/OpenPromise.js +12 -0
- package/dist/utilities/OpenPromise.js.map +1 -0
- package/dist/utilities/PackageManager.d.ts +4 -0
- package/dist/utilities/PackageManager.js +46 -0
- package/dist/utilities/PackageManager.js.map +1 -0
- package/dist/utilities/RandomGenerator.d.ts +35 -0
- package/dist/utilities/RandomGenerator.js +149 -0
- package/dist/utilities/RandomGenerator.js.map +1 -0
- package/dist/utilities/TarballMaker.d.ts +4 -0
- package/dist/utilities/TarballMaker.js +50 -0
- package/dist/utilities/TarballMaker.js.map +1 -0
- package/dist/utilities/Timeline.d.ts +20 -0
- package/dist/utilities/Timeline.js +110 -0
- package/dist/utilities/Timeline.js.map +1 -0
- package/dist/utilities/Timer.d.ts +13 -0
- package/dist/utilities/Timer.js +70 -0
- package/dist/utilities/Timer.js.map +1 -0
- package/dist/utilities/Utilities.d.ts +68 -0
- package/dist/utilities/Utilities.js +305 -0
- package/dist/utilities/Utilities.js.map +1 -0
- package/dist/utilities/WasmMemoryManager.d.ts +142 -0
- package/dist/utilities/WasmMemoryManager.js +407 -0
- package/dist/utilities/WasmMemoryManager.js.map +1 -0
- package/dist/utilities/WebReader.d.ts +1 -0
- package/dist/utilities/WebReader.js +47 -0
- package/dist/utilities/WebReader.js.map +1 -0
- package/dist/utilities/WikipediaReader.d.ts +1 -0
- package/dist/utilities/WikipediaReader.js +31 -0
- package/dist/utilities/WikipediaReader.js.map +1 -0
- package/dist/voice-activity-detection/SileroVAD.d.ts +13 -0
- package/dist/voice-activity-detection/SileroVAD.js +58 -0
- package/dist/voice-activity-detection/SileroVAD.js.map +1 -0
- package/dist/voice-activity-detection/WebRtcVAD.d.ts +3 -0
- package/dist/voice-activity-detection/WebRtcVAD.js +53 -0
- package/dist/voice-activity-detection/WebRtcVAD.js.map +1 -0
- package/docs/CLI.md +234 -0
- package/docs/Development.md +3 -0
- package/docs/Engines.md +80 -0
- package/docs/Licenses.md +37 -0
- package/docs/Options.md +192 -0
- package/docs/Roadmap.md +16 -0
- package/docs/Technical.md +72 -0
- package/package.json +119 -19
- package/cli.js +0 -3
|
@@ -0,0 +1,467 @@
|
|
|
1
|
+
export function ipaPhoneToKirshenbaum(ipaPhone) {
|
|
2
|
+
let result = '';
|
|
3
|
+
for (const char of ipaPhone) {
|
|
4
|
+
const convertedChar = ipaToKirshenbaum[char];
|
|
5
|
+
if (convertedChar == undefined) {
|
|
6
|
+
throw new Error(`Could not convert phone character '${char}' to Kirshenbaum encoding`);
|
|
7
|
+
}
|
|
8
|
+
result += convertedChar || '_';
|
|
9
|
+
}
|
|
10
|
+
return result;
|
|
11
|
+
}
|
|
12
|
+
export function ipaWordToTimitTokens(ipaWord, subphoneCount = 0) {
|
|
13
|
+
let result = [];
|
|
14
|
+
for (const ipaPhone of ipaWord) {
|
|
15
|
+
const convertedPhoneSequence = ipaPhoneToTimit(ipaPhone, subphoneCount);
|
|
16
|
+
if (convertedPhoneSequence == undefined) {
|
|
17
|
+
throw new Error(`Could not find a TIMIT equivalent for ipa phone '${ipaPhone}'`);
|
|
18
|
+
}
|
|
19
|
+
result = [...result, ...convertedPhoneSequence];
|
|
20
|
+
}
|
|
21
|
+
return result;
|
|
22
|
+
}
|
|
23
|
+
export function ipaPhoneToTimit(ipaPhone, subphoneCount = 0) {
|
|
24
|
+
let tokens = ipaToTimit[ipaPhone];
|
|
25
|
+
if (!tokens) {
|
|
26
|
+
return undefined;
|
|
27
|
+
}
|
|
28
|
+
if (subphoneCount > 0) {
|
|
29
|
+
tokens = splitTokensToSubphones(tokens, subphoneCount);
|
|
30
|
+
}
|
|
31
|
+
return tokens;
|
|
32
|
+
}
|
|
33
|
+
export function arpabetPhoneToIpa(arpabetPhone) {
|
|
34
|
+
return arpabetToIPA[arpabetPhone.toUpperCase()];
|
|
35
|
+
}
|
|
36
|
+
export function timitPhoneToIpa(arpabetPhone) {
|
|
37
|
+
return timitToIPA[arpabetPhone.toUpperCase()];
|
|
38
|
+
}
|
|
39
|
+
export function splitTokensToSubphones(tokens, subphoneCount = 3) {
|
|
40
|
+
const result = [];
|
|
41
|
+
for (const token of tokens) {
|
|
42
|
+
if (token == 'pause') {
|
|
43
|
+
result.push(token);
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
for (let i = 1; i <= subphoneCount; i++) {
|
|
47
|
+
result.push(`${token}/${i}`);
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
return result;
|
|
51
|
+
}
|
|
52
|
+
// ARPABET was invented for English.
|
|
53
|
+
// The standard dictionary written in ARPABET is the CMU dictionary.
|
|
54
|
+
// TIMIT is written in a variant of ARPABET that includes a couple
|
|
55
|
+
// of non-standard allophones, and most significantly, includes
|
|
56
|
+
// separate symbols for the closure and release portions of each stop.
|
|
57
|
+
const arpabetToIPA = {
|
|
58
|
+
'AA': 'ɑ',
|
|
59
|
+
'AE': 'æ',
|
|
60
|
+
'AH': 'ʌ',
|
|
61
|
+
'AH0': 'ə',
|
|
62
|
+
'AO': 'ɔ',
|
|
63
|
+
'AW': 'aʊ',
|
|
64
|
+
'AY': 'aɪ',
|
|
65
|
+
'EH': 'ɛ',
|
|
66
|
+
'ER': 'ɝ',
|
|
67
|
+
'ER0': 'ɚ',
|
|
68
|
+
'EY': 'eɪ',
|
|
69
|
+
'IH': 'ɪ',
|
|
70
|
+
'IH0': 'ɨ',
|
|
71
|
+
'IY': 'i',
|
|
72
|
+
'OW': 'oʊ',
|
|
73
|
+
'OY': 'ɔɪ',
|
|
74
|
+
'UH': 'ʊ',
|
|
75
|
+
'UW': 'u',
|
|
76
|
+
'B': 'b',
|
|
77
|
+
'CH': 'tʃ',
|
|
78
|
+
'D': 'd',
|
|
79
|
+
'DH': 'ð',
|
|
80
|
+
'EL': 'l̩ ',
|
|
81
|
+
'EM': 'm̩',
|
|
82
|
+
'EN': 'n̩',
|
|
83
|
+
'F': 'f',
|
|
84
|
+
'G': 'ɡ',
|
|
85
|
+
'HH': 'h',
|
|
86
|
+
'JH': 'dʒ',
|
|
87
|
+
'K': 'k',
|
|
88
|
+
'L': 'l',
|
|
89
|
+
'M': 'm',
|
|
90
|
+
'N': 'n',
|
|
91
|
+
'NG': 'ŋ',
|
|
92
|
+
'P': 'p',
|
|
93
|
+
'Q': 'ʔ',
|
|
94
|
+
'R': 'ɹ',
|
|
95
|
+
'S': 's',
|
|
96
|
+
'SH': 'ʃ',
|
|
97
|
+
'T': 't',
|
|
98
|
+
'TH': 'θ',
|
|
99
|
+
'V': 'v',
|
|
100
|
+
'W': 'w',
|
|
101
|
+
'WH': 'ʍ',
|
|
102
|
+
'Y': 'j',
|
|
103
|
+
'Z': 'z',
|
|
104
|
+
'ZH': 'ʒ'
|
|
105
|
+
};
|
|
106
|
+
const timitToIPA = {
|
|
107
|
+
...arpabetToIPA,
|
|
108
|
+
'AX': 'ə',
|
|
109
|
+
'AX-H': 'ə̥',
|
|
110
|
+
'AXR': 'ɚ',
|
|
111
|
+
'B': '',
|
|
112
|
+
'BCL': 'b',
|
|
113
|
+
'D': '',
|
|
114
|
+
'DCL': 'd',
|
|
115
|
+
'DX': 'ɾ',
|
|
116
|
+
'ENG': 'ŋ̍',
|
|
117
|
+
'EPI': '',
|
|
118
|
+
'G': '',
|
|
119
|
+
'GCL': 'g',
|
|
120
|
+
'HV': 'ɦ',
|
|
121
|
+
'H//': '',
|
|
122
|
+
'IX': 'ɨ',
|
|
123
|
+
'KCL': 'k',
|
|
124
|
+
'K': '',
|
|
125
|
+
'NX': 'ɾ̃',
|
|
126
|
+
'P': '',
|
|
127
|
+
'PAU': '',
|
|
128
|
+
'PCL': 'p',
|
|
129
|
+
'T': '',
|
|
130
|
+
'TCL': 't',
|
|
131
|
+
'UX': 'ʉ',
|
|
132
|
+
};
|
|
133
|
+
const ipaToTimit = {
|
|
134
|
+
// Vowels
|
|
135
|
+
'ɑ': ['aa'],
|
|
136
|
+
'ɑː': ['aa'],
|
|
137
|
+
'ɑːɹ': ['aa', 'r'],
|
|
138
|
+
'a': ['aa'],
|
|
139
|
+
'aʊ': ['aw'],
|
|
140
|
+
'aɪ': ['ay'],
|
|
141
|
+
'aɪɚ': ['ay', 'r'],
|
|
142
|
+
'aɪə': ['ay', 'ax'],
|
|
143
|
+
'æ': ['ae'],
|
|
144
|
+
'æʊ': ['ae', 'uh'],
|
|
145
|
+
'ʌ': ['ah'],
|
|
146
|
+
'ɐ': ['ah'],
|
|
147
|
+
'ə': ['ax'],
|
|
148
|
+
'ə̥': ['ax-h'],
|
|
149
|
+
'əʊ': ['ax', 'uh'],
|
|
150
|
+
'ɚ': ['axr'],
|
|
151
|
+
'ᵻ': ['ax'],
|
|
152
|
+
'ɔ': ['ao'],
|
|
153
|
+
'ɔː': ['ao'],
|
|
154
|
+
'ɔːɹ': ['ao', 'r'],
|
|
155
|
+
'ɔɪ': ['oy'],
|
|
156
|
+
'ɒ': ['ao'],
|
|
157
|
+
'ɛ': ['eh'],
|
|
158
|
+
'ɜː': ['eh'],
|
|
159
|
+
'ɝ': ['er'],
|
|
160
|
+
'ɛɹ': ['er'],
|
|
161
|
+
'eɪ': ['ey'],
|
|
162
|
+
'eə': ['ey', 'ax'],
|
|
163
|
+
'eː': ['ey'],
|
|
164
|
+
'ɪ': ['ih'],
|
|
165
|
+
'ɪɹ': ['ih', 'r'],
|
|
166
|
+
'ɨ': ['ix'],
|
|
167
|
+
'i': ['iy'],
|
|
168
|
+
'iː': ['iy'],
|
|
169
|
+
'iə': ['iy', 'ax'],
|
|
170
|
+
'oʊ': ['ow'],
|
|
171
|
+
'oː': ['ow'],
|
|
172
|
+
'oːɹ': ['ow', 'r'],
|
|
173
|
+
'ʊ': ['uh'],
|
|
174
|
+
'ʊɹ': ['uh', 'r'],
|
|
175
|
+
'ʊə': ['uh', 'ax'],
|
|
176
|
+
'u': ['uw'],
|
|
177
|
+
'uː': ['uw'],
|
|
178
|
+
// Consonants
|
|
179
|
+
'b': ['bcl', 'b'],
|
|
180
|
+
'tʃ': ['tcl', 'ch'],
|
|
181
|
+
'd': ['dcl', 'd'],
|
|
182
|
+
'ð': ['dh'],
|
|
183
|
+
'l̩': ['el'],
|
|
184
|
+
'əl': ['el'],
|
|
185
|
+
'm̩': ['em'],
|
|
186
|
+
'n̩': ['en'],
|
|
187
|
+
'f': ['f'],
|
|
188
|
+
'ɡ': ['gcl', 'g'],
|
|
189
|
+
'h': ['hh'],
|
|
190
|
+
'dʒ': ['dcl', 'jh'],
|
|
191
|
+
'k': ['kcl', 'k'],
|
|
192
|
+
'l': ['l'],
|
|
193
|
+
'm': ['m'],
|
|
194
|
+
'n': ['n'],
|
|
195
|
+
'ŋ': ['ng'],
|
|
196
|
+
'p': ['pcl', 'p'],
|
|
197
|
+
'ʔ': ['q'],
|
|
198
|
+
'ɹ': ['r'],
|
|
199
|
+
's': ['s'],
|
|
200
|
+
'ʃ': ['sh'],
|
|
201
|
+
't': ['tcl', 't'],
|
|
202
|
+
'θ': ['th'],
|
|
203
|
+
'v': ['v'],
|
|
204
|
+
'w': ['w'],
|
|
205
|
+
'ʍ': ['wh'],
|
|
206
|
+
'j': ['y'],
|
|
207
|
+
'z': ['z'],
|
|
208
|
+
'ʒ': ['zh'],
|
|
209
|
+
'ɾ̃': ['nx'],
|
|
210
|
+
'ʉ': ['ux'],
|
|
211
|
+
'ɦ': ['hv'],
|
|
212
|
+
'ŋ̍': ['eng'],
|
|
213
|
+
'ɾ': ['dx'],
|
|
214
|
+
};
|
|
215
|
+
// This is adapted from a lookup table on the eSpeak-ng source code
|
|
216
|
+
const ipaToKirshenbaum = {
|
|
217
|
+
'1': '1',
|
|
218
|
+
'2': '2',
|
|
219
|
+
'4': '4',
|
|
220
|
+
'5': '5',
|
|
221
|
+
'6': '6',
|
|
222
|
+
'7': '7',
|
|
223
|
+
'9': '9',
|
|
224
|
+
' ': ' ',
|
|
225
|
+
'!': '!',
|
|
226
|
+
'\'': '\'',
|
|
227
|
+
'ʰ': '#',
|
|
228
|
+
'$': '$',
|
|
229
|
+
'%': '%',
|
|
230
|
+
//'æ': '&',
|
|
231
|
+
'æ': 'a',
|
|
232
|
+
'ˈ': '\'',
|
|
233
|
+
'(': '(',
|
|
234
|
+
')': ')',
|
|
235
|
+
'ɾ': '*',
|
|
236
|
+
'+': '+',
|
|
237
|
+
'ˌ': ',',
|
|
238
|
+
'-': '-',
|
|
239
|
+
'.': '.',
|
|
240
|
+
'/': '/',
|
|
241
|
+
'ɒ': '0',
|
|
242
|
+
'ɜ': '3',
|
|
243
|
+
'ɵ': '8',
|
|
244
|
+
'ː': ':',
|
|
245
|
+
'ʲ': ';',
|
|
246
|
+
'<': '<',
|
|
247
|
+
'=': '=',
|
|
248
|
+
'>': '>',
|
|
249
|
+
'ʔ': '?',
|
|
250
|
+
'ə': '@',
|
|
251
|
+
'ɑ': 'A',
|
|
252
|
+
'β': 'B',
|
|
253
|
+
'ç': 'C',
|
|
254
|
+
'ð': 'D',
|
|
255
|
+
'ɛ': 'E',
|
|
256
|
+
'F': 'F',
|
|
257
|
+
'ɢ': 'G',
|
|
258
|
+
'ħ': 'H',
|
|
259
|
+
'ɪ': 'I',
|
|
260
|
+
'ɟ': 'J',
|
|
261
|
+
'K': 'K',
|
|
262
|
+
'ɫ': 'L',
|
|
263
|
+
'ɱ': 'M',
|
|
264
|
+
'ŋ': 'N',
|
|
265
|
+
'ɔ': 'O',
|
|
266
|
+
'Φ': 'P',
|
|
267
|
+
'ɣ': 'Q',
|
|
268
|
+
'ʀ': 'R',
|
|
269
|
+
'ʃ': 'S',
|
|
270
|
+
'θ': 'T',
|
|
271
|
+
'ʊ': 'U',
|
|
272
|
+
'ʌ': 'V',
|
|
273
|
+
'œ': 'W',
|
|
274
|
+
'χ': 'X',
|
|
275
|
+
'ø': 'Y',
|
|
276
|
+
'ʒ': 'Z',
|
|
277
|
+
'̪': '[',
|
|
278
|
+
'\\': '\\',
|
|
279
|
+
']': ']',
|
|
280
|
+
'^': '^',
|
|
281
|
+
'_': '_',
|
|
282
|
+
'`': '`',
|
|
283
|
+
'a': 'a',
|
|
284
|
+
'b': 'b',
|
|
285
|
+
'c': 'c',
|
|
286
|
+
'd': 'd',
|
|
287
|
+
'e': 'e',
|
|
288
|
+
'f': 'f',
|
|
289
|
+
'ɡ': 'g',
|
|
290
|
+
'h': 'h',
|
|
291
|
+
'i': 'i',
|
|
292
|
+
'j': 'j',
|
|
293
|
+
'k': 'k',
|
|
294
|
+
'l': 'l',
|
|
295
|
+
'm': 'm',
|
|
296
|
+
'n': 'n',
|
|
297
|
+
'o': 'o',
|
|
298
|
+
'p': 'p',
|
|
299
|
+
'q': 'q',
|
|
300
|
+
'r': 'r',
|
|
301
|
+
's': 's',
|
|
302
|
+
't': 't',
|
|
303
|
+
'u': 'u',
|
|
304
|
+
'v': 'v',
|
|
305
|
+
'w': 'w',
|
|
306
|
+
'x': 'x',
|
|
307
|
+
'y': 'y',
|
|
308
|
+
'z': 'z',
|
|
309
|
+
'{': '{',
|
|
310
|
+
'|': '|',
|
|
311
|
+
'}': '}',
|
|
312
|
+
'̃': '~',
|
|
313
|
+
'': '',
|
|
314
|
+
// Extensions
|
|
315
|
+
'ɚ': '3',
|
|
316
|
+
'ɹ': 'r',
|
|
317
|
+
'ɐ': 'a#',
|
|
318
|
+
'ᵻ': 'i',
|
|
319
|
+
'̩': ','
|
|
320
|
+
};
|
|
321
|
+
/*
|
|
322
|
+
// Source: https://github.com/coruus/ascii-ipa/blob/master/kirshenbaum.py
|
|
323
|
+
// Conversion regex: \('.*?': '.*?'\)
|
|
324
|
+
// Replace: $2: $1
|
|
325
|
+
const ipaToKirshenbaum2: { [p: string]: string | undefined } = {
|
|
326
|
+
'm': 'm',
|
|
327
|
+
'p': 'p',
|
|
328
|
+
'b': 'b',
|
|
329
|
+
'Φ': 'P',
|
|
330
|
+
'β': 'B',
|
|
331
|
+
'ʙ': 'b<trl>',
|
|
332
|
+
'pʼ': 'p`',
|
|
333
|
+
'ɓ': 'b`',
|
|
334
|
+
'ʘ': 'p!',
|
|
335
|
+
'ɱ': 'M',
|
|
336
|
+
'f': 'f',
|
|
337
|
+
'v': 'v',
|
|
338
|
+
'ʋ': 'r<lbd>',
|
|
339
|
+
'n\u032a': 'n[',
|
|
340
|
+
't\u032a': 't[',
|
|
341
|
+
'θ': 'T',
|
|
342
|
+
'ð': 'D',
|
|
343
|
+
'r\u032a': 'r[',
|
|
344
|
+
'l\u032a': 'l[',
|
|
345
|
+
't\u032a\u02bc': 't[`',
|
|
346
|
+
'ɗ': 'd`',
|
|
347
|
+
'ʇ': 't!',
|
|
348
|
+
'n': 'n',
|
|
349
|
+
't': 't',
|
|
350
|
+
'd': 'd',
|
|
351
|
+
's': 's',
|
|
352
|
+
'z': 'z',
|
|
353
|
+
'ɬ': 's<lat>',
|
|
354
|
+
'ɮ': 'z<lat>',
|
|
355
|
+
'ɹ': 'r',
|
|
356
|
+
'l': 'l',
|
|
357
|
+
'ʀ': 'r<trl>',
|
|
358
|
+
'ɾ': '*',
|
|
359
|
+
'ɺ': '*<lat>',
|
|
360
|
+
't\u02bc': 't`',
|
|
361
|
+
'ʗ': 'c!',
|
|
362
|
+
'ʖ': 'l!',
|
|
363
|
+
'ɳ': 'n.',
|
|
364
|
+
'ʈ': 't.',
|
|
365
|
+
'ɖ': 'd.',
|
|
366
|
+
'ʂ': 's.',
|
|
367
|
+
'ʐ': 'z.',
|
|
368
|
+
//'ɖ': 'r.',
|
|
369
|
+
'ɭ': 'l.',
|
|
370
|
+
'ɽ': '*.',
|
|
371
|
+
'ʃ': 'S',
|
|
372
|
+
'ʒ': 'Z',
|
|
373
|
+
'n^': 'n^',
|
|
374
|
+
'c': 'c',
|
|
375
|
+
'ɟ': 'J',
|
|
376
|
+
'ç': 'C',
|
|
377
|
+
'ʝ': 'C<vcd>',
|
|
378
|
+
'j': 'j',
|
|
379
|
+
'ɥ': 'j<rnd>',
|
|
380
|
+
'ʎ': 'l^',
|
|
381
|
+
'ʄ': 'J`',
|
|
382
|
+
'ŋ': 'N',
|
|
383
|
+
'k': 'k',
|
|
384
|
+
'g': 'g',
|
|
385
|
+
'x': 'x',
|
|
386
|
+
'ɣ': 'Q',
|
|
387
|
+
'ɰ': 'j<vel>',
|
|
388
|
+
'ɫ': 'L',
|
|
389
|
+
//'ɬ': '{vls,alv,lat,frc}',
|
|
390
|
+
'k\u02bc': 'k`',
|
|
391
|
+
'g\u02bc': 'g`',
|
|
392
|
+
'ʞ': 'k!',
|
|
393
|
+
'n\u2030g': 'n<lbv>',
|
|
394
|
+
'k\u2030p': 't<lbv>',
|
|
395
|
+
'g\u2030b': 'n<lbv>',
|
|
396
|
+
'ʍ': 'w<vls>',
|
|
397
|
+
'w': 'w',
|
|
398
|
+
'ɴ': 'n'',
|
|
399
|
+
'q': 'q',
|
|
400
|
+
'ɢ': 'G',
|
|
401
|
+
'χ': 'X',
|
|
402
|
+
'ʁ': 'g'',
|
|
403
|
+
//'ʀ': 'r'',
|
|
404
|
+
'ʠ': 'q`',
|
|
405
|
+
'ʛ': 'G`',
|
|
406
|
+
'ħ': 'H',
|
|
407
|
+
'ʕ': 'H<vcd>',
|
|
408
|
+
'ʔ': '?',
|
|
409
|
+
'h': 'h',
|
|
410
|
+
'ɦ': 'h<?>',
|
|
411
|
+
'i': 'i',
|
|
412
|
+
'y': 'y',
|
|
413
|
+
'ɪ': 'I',
|
|
414
|
+
'ʏ': 'I.',
|
|
415
|
+
'e': 'e',
|
|
416
|
+
'ø': 'Y',
|
|
417
|
+
'ɛ': 'E',
|
|
418
|
+
'œ': 'W',
|
|
419
|
+
'æ': '&',
|
|
420
|
+
'ɶ': '&.',
|
|
421
|
+
'ɨ': 'i'',
|
|
422
|
+
'ʉ': 'u'',
|
|
423
|
+
'ɘ': '@<umd>',
|
|
424
|
+
'ɝ': 'R<umd>',
|
|
425
|
+
'ə': '@',
|
|
426
|
+
'ɚ': 'R',
|
|
427
|
+
'ɵ': '@.',
|
|
428
|
+
//'ɜ': 'V'',
|
|
429
|
+
'ɜ': '3',
|
|
430
|
+
'ɞ': 'O'',
|
|
431
|
+
'a': 'a',
|
|
432
|
+
'ɯ': 'u-',
|
|
433
|
+
'u': 'u',
|
|
434
|
+
'ʊ': 'U',
|
|
435
|
+
'ɤ': 'o-',
|
|
436
|
+
'o': 'o',
|
|
437
|
+
'ʌ': 'V',
|
|
438
|
+
'ɔ': 'O',
|
|
439
|
+
'ɑ': 'A',
|
|
440
|
+
'ɒ': 'A.',
|
|
441
|
+
|
|
442
|
+
'ː': ':',
|
|
443
|
+
'\u0322': '.',
|
|
444
|
+
'\u02bc': '`',
|
|
445
|
+
'\u032a': '[',
|
|
446
|
+
'\u02b2': ';',
|
|
447
|
+
''': ''',
|
|
448
|
+
'^': '^',
|
|
449
|
+
'\u0334': '<H>',
|
|
450
|
+
'\u02b0': '<h>',
|
|
451
|
+
'\u02da': '<unx>',
|
|
452
|
+
'\u0325': '<vls>',
|
|
453
|
+
//'\u02da': '<o>',
|
|
454
|
+
'\u02b3': '<r>',
|
|
455
|
+
'\u02b7': '<w>',
|
|
456
|
+
'\u02b1': '<?>',
|
|
457
|
+
|
|
458
|
+
'\u0303': '~',
|
|
459
|
+
//'\u0334': '~'
|
|
460
|
+
|
|
461
|
+
'ˈ': ''',
|
|
462
|
+
'ˌ': ',',
|
|
463
|
+
' ': ' ',
|
|
464
|
+
'\n': '\n'
|
|
465
|
+
}
|
|
466
|
+
*/
|
|
467
|
+
//# sourceMappingURL=PhoneConversion.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PhoneConversion.js","sourceRoot":"","sources":["../../src/nlp/PhoneConversion.ts"],"names":[],"mappings":"AAAA,MAAM,UAAU,qBAAqB,CAAC,QAAgB;IACrD,IAAI,MAAM,GAAG,EAAE,CAAA;IAEf,KAAK,MAAM,IAAI,IAAI,QAAQ,EAAE;QAC5B,MAAM,aAAa,GAAG,gBAAgB,CAAC,IAAI,CAAC,CAAA;QAE5C,IAAI,aAAa,IAAI,SAAS,EAAE;YAC/B,MAAM,IAAI,KAAK,CAAC,sCAAsC,IAAI,2BAA2B,CAAC,CAAA;SACtF;QAED,MAAM,IAAI,aAAa,IAAI,GAAG,CAAA;KAC9B;IAED,OAAO,MAAM,CAAA;AACd,CAAC;AAED,MAAM,UAAU,oBAAoB,CAAC,OAAe,EAAE,aAAa,GAAG,CAAC;IACtE,IAAI,MAAM,GAAa,EAAE,CAAA;IAEzB,KAAK,MAAM,QAAQ,IAAI,OAAO,EAAE;QAC/B,MAAM,sBAAsB,GAAG,eAAe,CAAC,QAAQ,EAAE,aAAa,CAAC,CAAA;QAEvE,IAAI,sBAAsB,IAAI,SAAS,EAAE;YACxC,MAAM,IAAI,KAAK,CAAC,oDAAoD,QAAQ,GAAG,CAAC,CAAA;SAChF;QAED,MAAM,GAAG,CAAC,GAAG,MAAM,EAAE,GAAG,sBAAsB,CAAC,CAAA;KAC/C;IAED,OAAO,MAAM,CAAA;AACd,CAAC;AAED,MAAM,UAAU,eAAe,CAAC,QAAgB,EAAE,aAAa,GAAG,CAAC;IAClE,IAAI,MAAM,GAAG,UAAU,CAAC,QAAQ,CAAC,CAAA;IAEjC,IAAI,CAAC,MAAM,EAAE;QACZ,OAAO,SAAS,CAAA;KAChB;IAED,IAAI,aAAa,GAAG,CAAC,EAAE;QACtB,MAAM,GAAG,sBAAsB,CAAC,MAAM,EAAE,aAAa,CAAC,CAAA;KACtD;IAED,OAAO,MAAM,CAAA;AACd,CAAC;AAED,MAAM,UAAU,iBAAiB,CAAC,YAAoB;IACrD,OAAO,YAAY,CAAC,YAAY,CAAC,WAAW,EAAE,CAAC,CAAA;AAChD,CAAC;AAED,MAAM,UAAU,eAAe,CAAC,YAAoB;IACnD,OAAO,UAAU,CAAC,YAAY,CAAC,WAAW,EAAE,CAAC,CAAA;AAC9C,CAAC;AAED,MAAM,UAAU,sBAAsB,CAAC,MAAgB,EAAE,aAAa,GAAG,CAAC;IACzE,MAAM,MAAM,GAAa,EAAE,CAAA;IAE3B,KAAK,MAAM,KAAK,IAAI,MAAM,EAAE;QAC3B,IAAI,KAAK,IAAI,OAAO,EAAE;YACrB,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAA;YAElB,SAAQ;SACR;QAED,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,aAAa,EAAE,CAAC,EAAE,EAAE;YACxC,MAAM,CAAC,IAAI,CAAC,GAAG,KAAK,IAAI,CAAC,EAAE,CAAC,CAAA;SAC5B;KACD;IAED,OAAO,MAAM,CAAA;AACd,CAAC;AAED,oCAAoC;AACpC,oEAAoE;AACpE,kEAAkE;AAClE,+DAA+D;AAC/D,sEAAsE;AACtE,MAAM,YAAY,GAAwC;IACzD,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,GAAG;IACT,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,KAAK;IACX,IAAI,EAAE,IAAI;IACV,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;IACT,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,GAAG;CACT,CAAA;AAED,MAAM,UAAU,GAAwC;IACvD,GAAG,YAAY;IAEf,IAAI,EAAE,GAAG;IACT,MAAM,EAAE,IAAI;IACZ,KAAK,EAAE,GAAG;IACV,GAAG,EAAE,EAAE;IACP,KAAK,EAAE,GAAG;IACV,GAAG,EAAE,EAAE;IACP,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,IAAI;IACX,KAAK,EAAE,EAAE;IACT,GAAG,EAAE,EAAE;IACP,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,EAAE;IACT,IAAI,EAAE,GAAG;IACT,KAAK,EAAE,GAAG;IACV,GAAG,EAAE,EAAE;IACP,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,EAAE;IACP,KAAK,EAAE,EAAE;IACT,KAAK,EAAE,GAAG;IACV,GAAG,EAAE,EAAE;IACP,KAAK,EAAE,GAAG;IACV,IAAI,EAAE,GAAG;CACT,CAAA;AAED,MAAM,UAAU,GAA0C;IACzD,SAAS;IACT,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,KAAK,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IAClB,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,KAAK,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IAClB,KAAK,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAEnB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAElB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,MAAM,CAAC;IACd,IAAI,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAClB,GAAG,EAAE,CAAC,KAAK,CAAC;IACZ,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,KAAK,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IAClB,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IAEZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAClB,IAAI,EAAE,CAAC,IAAI,CAAC;IAEZ,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IAEjB,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAElB,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,KAAK,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IAElB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;IACjB,IAAI,EAAE,CAAC,IAAI,EAAE,IAAI,CAAC;IAElB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IAEZ,aAAa;IACb,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,IAAI,EAAE,CAAC,KAAK,EAAE,IAAI,CAAC;IACnB,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,KAAK,EAAE,IAAI,CAAC;IACnB,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC;IACjB,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,GAAG,CAAC;IACV,GAAG,EAAE,CAAC,IAAI,CAAC;IAEX,IAAI,EAAE,CAAC,IAAI,CAAC;IACZ,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,GAAG,EAAE,CAAC,IAAI,CAAC;IACX,IAAI,EAAE,CAAC,KAAK,CAAC;IACb,GAAG,EAAE,CAAC,IAAI,CAAC;CACX,CAAA;AAED,mEAAmE;AACnE,MAAM,gBAAgB,GAAwC;IAC1D,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,WAAW;IACd,GAAG,EAAE,GAAG;IACL,GAAG,EAAE,IAAI;IACT,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,IAAI,EAAE,IAAI;IACV,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACX,GAAG,EAAE,GAAG;IACL,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IAEX,aAAa;IACb,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,IAAI;IACT,GAAG,EAAE,GAAG;IACR,GAAG,EAAE,GAAG;CACR,CAAA;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAiJE"}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
export declare const wordCharacterPattern: RegExp;
|
|
2
|
+
export declare const phraseSeparators: string[];
|
|
3
|
+
export declare const sentenceSeparators: string[];
|
|
4
|
+
export declare const symbolWords: string[];
|
|
5
|
+
export declare function isWord(str: string): boolean;
|
|
6
|
+
export declare class Sentence {
|
|
7
|
+
phrases: Phrase[];
|
|
8
|
+
readonly isSentenceFinalizer = true;
|
|
9
|
+
get length(): number;
|
|
10
|
+
get text(): string;
|
|
11
|
+
}
|
|
12
|
+
export declare class Phrase {
|
|
13
|
+
words: Word[];
|
|
14
|
+
get length(): number;
|
|
15
|
+
get text(): string;
|
|
16
|
+
get lastWord(): Word | undefined;
|
|
17
|
+
get isSentenceFinalizer(): boolean;
|
|
18
|
+
}
|
|
19
|
+
export declare class Word {
|
|
20
|
+
readonly text: string;
|
|
21
|
+
isSentenceFinalizer: boolean;
|
|
22
|
+
constructor(text: string, isSentenceFinalizer: boolean);
|
|
23
|
+
get containsOnlyPunctuation(): boolean;
|
|
24
|
+
get isSymbolWord(): boolean;
|
|
25
|
+
get isPhraseSeperator(): boolean;
|
|
26
|
+
get length(): number;
|
|
27
|
+
}
|
|
28
|
+
export type Segment = Sentence | Phrase | Word;
|
|
29
|
+
export declare class Fragment {
|
|
30
|
+
segments: Segment[];
|
|
31
|
+
get length(): number;
|
|
32
|
+
get text(): string;
|
|
33
|
+
get isEmpty(): boolean;
|
|
34
|
+
get isNonempty(): boolean;
|
|
35
|
+
get lastSegment(): Segment | undefined;
|
|
36
|
+
}
|
|
37
|
+
export declare function splitToFragments(text: string, maxFragmentLength: number, langCode: string, preserveSentences?: boolean, preservePhrases?: boolean): Promise<Fragment[]>;
|
|
38
|
+
export declare function parse(text: string, langCode: string): Promise<Sentence[]>;
|
|
39
|
+
export declare function splitToSentences(text: string, langCode: string): string[];
|
|
40
|
+
export declare function splitToWords(text: string, langCode: string): Promise<string[]>;
|
|
41
|
+
export declare function splitToParagraphs(text: string): string[];
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
import XRegExp from 'xregexp';
|
|
2
|
+
import * as CldrSegmentation from 'cldr-segmentation';
|
|
3
|
+
import { splitChineseTextToWords_Jieba } from './ChineseSegmentation.js';
|
|
4
|
+
import { sumArray, includesAnyOf, indexOfAnyOf, logToStderr } from '../utilities/Utilities.js';
|
|
5
|
+
import { getShortLanguageCode } from '../utilities/Locale.js';
|
|
6
|
+
import { splitJapaneseTextToWords_Kuromoji } from './JapaneseSegmentation.js';
|
|
7
|
+
const log = logToStderr;
|
|
8
|
+
export const wordCharacterPattern = XRegExp('[\\p{Letter}\\p{Number}]');
|
|
9
|
+
export const phraseSeparators = [",", ";", ":"];
|
|
10
|
+
export const sentenceSeparators = [".", "?", "!"];
|
|
11
|
+
export const symbolWords = ["$", "%", "&", "@", "+", "/", "*", "="];
|
|
12
|
+
export function isWord(str) {
|
|
13
|
+
str = str.trim();
|
|
14
|
+
return wordCharacterPattern.test(str) || symbolWords.includes(str);
|
|
15
|
+
}
|
|
16
|
+
export class Sentence {
|
|
17
|
+
phrases = [];
|
|
18
|
+
isSentenceFinalizer = true;
|
|
19
|
+
get length() { return sumArray(this.phrases, (phrase) => phrase.length); }
|
|
20
|
+
get text() { return this.phrases.reduce((result, phrase) => result + phrase.text, ""); }
|
|
21
|
+
}
|
|
22
|
+
export class Phrase {
|
|
23
|
+
words = [];
|
|
24
|
+
get length() { return sumArray(this.words, (word) => word.length); }
|
|
25
|
+
get text() { return this.words.reduce((result, word) => result + word.text, ""); }
|
|
26
|
+
get lastWord() {
|
|
27
|
+
if (this.words.length == 0) {
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
return this.words[this.words.length - 1];
|
|
31
|
+
}
|
|
32
|
+
get isSentenceFinalizer() { return this.lastWord != null ? this.lastWord.isSentenceFinalizer : false; }
|
|
33
|
+
}
|
|
34
|
+
export class Word {
|
|
35
|
+
text;
|
|
36
|
+
isSentenceFinalizer;
|
|
37
|
+
constructor(text, isSentenceFinalizer) {
|
|
38
|
+
this.text = text;
|
|
39
|
+
this.isSentenceFinalizer = isSentenceFinalizer;
|
|
40
|
+
}
|
|
41
|
+
get containsOnlyPunctuation() { return !wordCharacterPattern.test(this.text) && !this.isSymbolWord; }
|
|
42
|
+
get isSymbolWord() { return symbolWords.includes(this.text); }
|
|
43
|
+
get isPhraseSeperator() { return this.containsOnlyPunctuation && includesAnyOf(this.text, phraseSeparators); }
|
|
44
|
+
get length() { return this.text.length; }
|
|
45
|
+
}
|
|
46
|
+
export class Fragment {
|
|
47
|
+
segments = [];
|
|
48
|
+
get length() { return sumArray(this.segments, (phrase) => phrase.length); }
|
|
49
|
+
get text() { return this.segments.reduce((result, segment) => result + segment.text, ""); }
|
|
50
|
+
get isEmpty() { return this.length == 0; }
|
|
51
|
+
get isNonempty() { return !this.isEmpty; }
|
|
52
|
+
get lastSegment() {
|
|
53
|
+
if (this.isEmpty) {
|
|
54
|
+
return undefined;
|
|
55
|
+
}
|
|
56
|
+
return this.segments[this.segments.length - 1];
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
export async function splitToFragments(text, maxFragmentLength, langCode, preserveSentences = true, preservePhrases = true) {
|
|
60
|
+
const parsedText = await parse(text, langCode);
|
|
61
|
+
const fragments = [];
|
|
62
|
+
let currentFragment = new Fragment();
|
|
63
|
+
const remainingCharactersInCurrentFragment = () => maxFragmentLength - currentFragment.length;
|
|
64
|
+
const createNewFragmentIfNeeded = () => {
|
|
65
|
+
if (currentFragment.isNonempty) {
|
|
66
|
+
fragments.push(currentFragment);
|
|
67
|
+
currentFragment = new Fragment();
|
|
68
|
+
}
|
|
69
|
+
};
|
|
70
|
+
const fitsCurrentFragment = (segment) => segment.length <= remainingCharactersInCurrentFragment();
|
|
71
|
+
for (const sentence of parsedText) {
|
|
72
|
+
if (fitsCurrentFragment(sentence)) {
|
|
73
|
+
currentFragment.segments.push(sentence);
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
if (preserveSentences) {
|
|
77
|
+
createNewFragmentIfNeeded();
|
|
78
|
+
if (fitsCurrentFragment(sentence)) {
|
|
79
|
+
currentFragment.segments.push(sentence);
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
for (const phrase of sentence.phrases) {
|
|
84
|
+
if (fitsCurrentFragment(phrase)) {
|
|
85
|
+
currentFragment.segments.push(phrase);
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
if (preservePhrases) {
|
|
89
|
+
createNewFragmentIfNeeded();
|
|
90
|
+
if (fitsCurrentFragment(phrase)) {
|
|
91
|
+
currentFragment.segments.push(phrase);
|
|
92
|
+
continue;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
for (const word of phrase.words) {
|
|
96
|
+
if (fitsCurrentFragment(word)) {
|
|
97
|
+
currentFragment.segments.push(word);
|
|
98
|
+
continue;
|
|
99
|
+
}
|
|
100
|
+
createNewFragmentIfNeeded();
|
|
101
|
+
if (fitsCurrentFragment(word)) {
|
|
102
|
+
currentFragment.segments.push(word);
|
|
103
|
+
continue;
|
|
104
|
+
}
|
|
105
|
+
throw new Error(`Encountered a word of length ${word.length}, which excceeds the maximum fragment length of ${maxFragmentLength}`);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
createNewFragmentIfNeeded();
|
|
110
|
+
return fragments;
|
|
111
|
+
}
|
|
112
|
+
export async function parse(text, langCode) {
|
|
113
|
+
const sentencesText = splitToSentences(text, langCode);
|
|
114
|
+
const sentences = [];
|
|
115
|
+
for (const sentenceText of sentencesText) {
|
|
116
|
+
const sentence = new Sentence();
|
|
117
|
+
let currentPhrase = new Phrase();
|
|
118
|
+
const wordTexts = await splitToWords(sentenceText, langCode);
|
|
119
|
+
for (let wordIndex = 0; wordIndex < wordTexts.length; wordIndex++) {
|
|
120
|
+
const word = new Word(wordTexts[wordIndex], wordIndex == wordTexts.length - 1);
|
|
121
|
+
if (word.isPhraseSeperator) {
|
|
122
|
+
const separatorIndex = indexOfAnyOf(word.text, phraseSeparators);
|
|
123
|
+
currentPhrase.words.push(new Word(word.text.substring(0, separatorIndex + 1), word.isSentenceFinalizer));
|
|
124
|
+
sentence.phrases.push(currentPhrase);
|
|
125
|
+
currentPhrase = new Phrase();
|
|
126
|
+
currentPhrase.words.push(new Word(word.text.substring(separatorIndex + 1), false));
|
|
127
|
+
}
|
|
128
|
+
else {
|
|
129
|
+
currentPhrase.words.push(word);
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
if (currentPhrase.words.length > 0) {
|
|
133
|
+
sentence.phrases.push(currentPhrase);
|
|
134
|
+
}
|
|
135
|
+
sentences.push(sentence);
|
|
136
|
+
}
|
|
137
|
+
return sentences;
|
|
138
|
+
}
|
|
139
|
+
export function splitToSentences(text, langCode) {
|
|
140
|
+
const shortLangCode = getShortLanguageCode(langCode || "");
|
|
141
|
+
return CldrSegmentation.sentenceSplit(text, CldrSegmentation.suppressions[shortLangCode]);
|
|
142
|
+
}
|
|
143
|
+
export async function splitToWords(text, langCode) {
|
|
144
|
+
const shortLangCode = getShortLanguageCode(langCode || "");
|
|
145
|
+
if (shortLangCode == "zh" || shortLangCode == "cmn") {
|
|
146
|
+
return splitChineseTextToWords_Jieba(text, true, true);
|
|
147
|
+
}
|
|
148
|
+
else if (shortLangCode == "ja") {
|
|
149
|
+
return splitJapaneseTextToWords_Kuromoji(text);
|
|
150
|
+
}
|
|
151
|
+
else {
|
|
152
|
+
return CldrSegmentation.wordSplit(text, CldrSegmentation.suppressions[shortLangCode]);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
export function splitToParagraphs(text) {
|
|
156
|
+
return text.split(/(\r?\n)+/g).map(p => p.trim()).filter(p => p.length > 0);
|
|
157
|
+
}
|
|
158
|
+
//# sourceMappingURL=Segmentation.js.map
|