echogarden 0.11.12 → 0.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/schemas/options.json +16 -0
- package/dist/api/Alignment.js +2 -2
- package/dist/api/Alignment.js.map +1 -1
- package/dist/api/Recognition.js +2 -2
- package/dist/api/Recognition.js.map +1 -1
- package/dist/api/Synthesis.js +5 -4
- package/dist/api/Synthesis.js.map +1 -1
- package/dist/api/Translation.js +2 -2
- package/dist/api/Translation.js.map +1 -1
- package/dist/audio/AudioUtilities.d.ts +1 -0
- package/dist/audio/AudioUtilities.js +25 -7
- package/dist/audio/AudioUtilities.js.map +1 -1
- package/dist/cli/CLI.js +2 -2
- package/dist/cli/CLI.js.map +1 -1
- package/dist/recognition/WhisperSTT.js +2 -2
- package/dist/recognition/WhisperSTT.js.map +1 -1
- package/dist/subtitles/Subtitles.d.ts +10 -7
- package/dist/subtitles/Subtitles.js +268 -207
- package/dist/subtitles/Subtitles.js.map +1 -1
- package/docs/Options.md +4 -2
- package/package.json +7 -6
- package/src/alignment/DTWMfccSequenceAlignment.ts +43 -0
- package/src/alignment/DTWSequenceAlignment.ts +121 -0
- package/src/alignment/DTWSequenceAlignmentWindowed.ts +210 -0
- package/src/alignment/LevenshteinSequenceAlignment.ts +126 -0
- package/src/alignment/SpeechAlignment.ts +488 -0
- package/src/api/API.ts +12 -0
- package/src/api/APIOptions.ts +15 -0
- package/src/api/Alignment.ts +329 -0
- package/src/api/Common.ts +16 -0
- package/src/api/Denoising.ts +120 -0
- package/src/api/LanguageDetection.ts +286 -0
- package/src/api/Recognition.ts +344 -0
- package/src/api/Synthesis.ts +1735 -0
- package/src/api/Translation.ts +143 -0
- package/src/api/Vad.ts +172 -0
- package/src/audio/AudioBufferConversion.ts +248 -0
- package/src/audio/AudioPlayer.ts +358 -0
- package/src/audio/AudioRecorder.ts +91 -0
- package/src/audio/AudioUtilities.ts +392 -0
- package/src/audio/SoxPath.ts +24 -0
- package/src/cli/CLI.ts +1360 -0
- package/src/cli/CLIConfigFile.ts +91 -0
- package/src/cli/CLILauncher.ts +26 -0
- package/src/cli/CLIOptionsSchema.ts +54 -0
- package/src/cli/CLIParser.ts +41 -0
- package/src/cli/CLIStarter.ts +40 -0
- package/src/codecs/FFMpegTranscoder.ts +214 -0
- package/src/codecs/TIMITCodec.ts +17 -0
- package/src/codecs/WaveCodec.ts +260 -0
- package/src/denoising/RNNoise.ts +95 -0
- package/src/dsp/BiquadFilter.ts +488 -0
- package/src/dsp/FFT.ts +187 -0
- package/src/dsp/MFCC.ts +227 -0
- package/src/dsp/MelSpectogram.ts +145 -0
- package/src/dsp/Rubberband.ts +249 -0
- package/src/dsp/Sonic.ts +59 -0
- package/src/dsp/SpeexResampler.ts +79 -0
- package/src/math/VectorMath.ts +812 -0
- package/src/nlp/ChineseSegmentation.ts +68 -0
- package/src/nlp/CompromiseNLP.ts +113 -0
- package/src/nlp/EspeakPhonemizer.ts +168 -0
- package/src/nlp/IPA.ts +139 -0
- package/src/nlp/JapaneseSegmentation.ts +53 -0
- package/src/nlp/Lexicon.ts +119 -0
- package/src/nlp/PhoneConversion.ts +508 -0
- package/src/nlp/Segmentation.ts +237 -0
- package/src/nlp/TextNormalizer.ts +160 -0
- package/src/recognition/AmazonTranscribeSTT.ts +112 -0
- package/src/recognition/AzureCognitiveServicesSTT.ts +76 -0
- package/src/recognition/GoogleCloudSTT.ts +92 -0
- package/src/recognition/SileroSTT.ts +173 -0
- package/src/recognition/VoskSTT.ts +112 -0
- package/src/recognition/WhisperSTT.ts +1518 -0
- package/src/server/Client.ts +297 -0
- package/src/server/Server.ts +178 -0
- package/src/server/ServerStarter.ts +12 -0
- package/src/server/Worker.ts +400 -0
- package/src/server/WorkerStarter.ts +38 -0
- package/src/speech-language-detection/SileroLanguageDetection.ts +105 -0
- package/src/subtitles/Subtitles.ts +478 -0
- package/src/synthesis/AwsPollyTTS.ts +78 -0
- package/src/synthesis/AzureCognitiveServicesTTS.ts +146 -0
- package/src/synthesis/CoquiServerTTS.ts +29 -0
- package/src/synthesis/ElevenLabsTTS.ts +104 -0
- package/src/synthesis/EspeakTTS.ts +552 -0
- package/src/synthesis/FliteTTS.ts +387 -0
- package/src/synthesis/GoogleCloudTTS.ts +112 -0
- package/src/synthesis/GoogleTranslateTTS.ts +210 -0
- package/src/synthesis/MicrosoftEdgeTTS.ts +298 -0
- package/src/synthesis/SamTTS.ts +30 -0
- package/src/synthesis/SapiTTS.ts +222 -0
- package/src/synthesis/StreamlabsPollyTTS.ts +114 -0
- package/src/synthesis/SvoxPicoTTS.ts +318 -0
- package/src/synthesis/VitsTTS.ts +734 -0
- package/src/tests/Test.ts +24 -0
- package/src/text-language-detection/FastTextLanguageDetection.ts +53 -0
- package/src/text-language-detection/TinyLDLanguageDetection.ts +16 -0
- package/src/typings/Fillers.d.ts +41 -0
- package/src/utilities/BinaryArrayConversion.ts +159 -0
- package/src/utilities/Compression.ts +91 -0
- package/src/utilities/FileDownloader.ts +201 -0
- package/src/utilities/FileSystem.ts +265 -0
- package/src/utilities/Hashing.ts +230 -0
- package/src/utilities/Locale.ts +119 -0
- package/src/utilities/Logger.ts +72 -0
- package/src/utilities/NdArrayUtilities.ts +31 -0
- package/src/utilities/ObjectUtilities.ts +169 -0
- package/src/utilities/OpenPromise.ts +13 -0
- package/src/utilities/PackageManager.ts +97 -0
- package/src/utilities/Queue.ts +17 -0
- package/src/utilities/RandomGenerator.ts +237 -0
- package/src/utilities/SignalChannel.ts +22 -0
- package/src/utilities/TarballMaker.ts +68 -0
- package/src/utilities/Timeline.ts +231 -0
- package/src/utilities/Timer.ts +93 -0
- package/src/utilities/Utilities.ts +574 -0
- package/src/utilities/WasmMemoryManager.ts +516 -0
- package/src/utilities/WebReader.ts +55 -0
- package/src/utilities/WikipediaReader.ts +41 -0
- package/src/voice-activity-detection/SileroVAD.ts +86 -0
- package/src/voice-activity-detection/WebRtcVAD.ts +76 -0
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
export async function splitChineseTextToWords_Jieba(text: string, fineGrained = false, useHMM = true) {
|
|
2
|
+
const jieba = await getWasmInstance()
|
|
3
|
+
|
|
4
|
+
if (!fineGrained) {
|
|
5
|
+
return jieba.cut(text, useHMM)
|
|
6
|
+
} else {
|
|
7
|
+
const results = jieba.tokenize(text, "search", useHMM)
|
|
8
|
+
|
|
9
|
+
const startOffsetsSet = new Set<number>()
|
|
10
|
+
const endOffsetsSet = new Set<number>()
|
|
11
|
+
|
|
12
|
+
for (const result of results) {
|
|
13
|
+
startOffsetsSet.add(result.start)
|
|
14
|
+
endOffsetsSet.add(result.end)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
const startOffsets = Array.from(startOffsetsSet)
|
|
18
|
+
startOffsets.sort((a, b) => a - b)
|
|
19
|
+
|
|
20
|
+
const endOffsets = Array.from(endOffsetsSet)
|
|
21
|
+
endOffsets.sort((a, b) => a - b)
|
|
22
|
+
|
|
23
|
+
const words: string[] = []
|
|
24
|
+
|
|
25
|
+
for (let i = 0; i < startOffsets.length; i++) {
|
|
26
|
+
const wordStartOffset = startOffsets[i]
|
|
27
|
+
|
|
28
|
+
function getWordEndOffset() {
|
|
29
|
+
if (i < startOffsets.length - 1) {
|
|
30
|
+
const nextWordStartOffset = startOffsets[i + 1]
|
|
31
|
+
|
|
32
|
+
for (let j = 0; j < endOffsets.length - 1; j++) {
|
|
33
|
+
const currentEndOffset = endOffsets[j]
|
|
34
|
+
const nextEndOffset = endOffsets[j + 1]
|
|
35
|
+
|
|
36
|
+
if (currentEndOffset >= nextWordStartOffset) {
|
|
37
|
+
return nextWordStartOffset
|
|
38
|
+
} else if (
|
|
39
|
+
currentEndOffset > wordStartOffset &&
|
|
40
|
+
currentEndOffset < nextWordStartOffset &&
|
|
41
|
+
nextEndOffset > nextWordStartOffset) {
|
|
42
|
+
|
|
43
|
+
return currentEndOffset
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
return endOffsets[endOffsets.length - 1]
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const wordEndOffset = getWordEndOffset()
|
|
52
|
+
|
|
53
|
+
words.push(text.substring(wordStartOffset, wordEndOffset))
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
return words
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
let JiebaWasmInstance: typeof import("jieba-wasm")
|
|
61
|
+
async function getWasmInstance() {
|
|
62
|
+
if (!JiebaWasmInstance) {
|
|
63
|
+
const { default: JibeaWasm } = await import("jieba-wasm")
|
|
64
|
+
JiebaWasmInstance = JibeaWasm
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
return JiebaWasmInstance
|
|
68
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
import { getShortLanguageCode } from '../utilities/Locale.js'
|
|
2
|
+
import { logToStderr } from '../utilities/Utilities.js'
|
|
3
|
+
import { Lexicon } from './Lexicon.js'
|
|
4
|
+
|
|
5
|
+
const log = logToStderr
|
|
6
|
+
|
|
7
|
+
export async function parse(text: string): Promise<CompromiseParsedDocument> {
|
|
8
|
+
const { default: nlp } = await import('compromise')
|
|
9
|
+
|
|
10
|
+
const doc = nlp(text)
|
|
11
|
+
|
|
12
|
+
doc.compute('penn')
|
|
13
|
+
|
|
14
|
+
const jsonDoc: any[] = doc.json({ offset: true })
|
|
15
|
+
|
|
16
|
+
//log(jsonDoc)
|
|
17
|
+
|
|
18
|
+
const result: CompromiseParsedDocument = jsonDoc.map(sentence => {
|
|
19
|
+
const terms = sentence.terms
|
|
20
|
+
const parsedSentence: CompromiseParsedSentence = []
|
|
21
|
+
|
|
22
|
+
for (let termIndex = 0; termIndex < terms.length; termIndex++) {
|
|
23
|
+
const term = terms[termIndex]
|
|
24
|
+
|
|
25
|
+
const parsedTerm: CompromiseParsedTerm = {
|
|
26
|
+
text: term.text,
|
|
27
|
+
pos: term.penn,
|
|
28
|
+
tags: term.tags,
|
|
29
|
+
preText: term.pre,
|
|
30
|
+
postText: term.post,
|
|
31
|
+
startOffset: term.offset.start,
|
|
32
|
+
endOffset: term.offset.start + term.offset.length
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
if (parsedTerm.text == "") {
|
|
36
|
+
if (parsedSentence.length > 0) {
|
|
37
|
+
parsedSentence[parsedSentence.length - 1].postText += parsedTerm.preText + parsedTerm.postText
|
|
38
|
+
}
|
|
39
|
+
} else if (parsedTerm.tags.includes("Abbreviation") && parsedTerm.postText.startsWith(".")) {
|
|
40
|
+
parsedTerm.text += "."
|
|
41
|
+
parsedTerm.endOffset += 1
|
|
42
|
+
parsedSentence.push(parsedTerm)
|
|
43
|
+
} else {
|
|
44
|
+
parsedSentence.push(parsedTerm)
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
return parsedSentence
|
|
49
|
+
})
|
|
50
|
+
|
|
51
|
+
//log(result)
|
|
52
|
+
|
|
53
|
+
return result
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export function tryMatchInLexicons(term: CompromiseParsedTerm, lexicons: Lexicon[], espeakVoice: string) {
|
|
57
|
+
const reversedLexicons = [...lexicons].reverse() // Give precedence to later lexicons
|
|
58
|
+
|
|
59
|
+
for (const lexicon of reversedLexicons) {
|
|
60
|
+
const match = tryMatchInLexicon(term, lexicon, espeakVoice)
|
|
61
|
+
|
|
62
|
+
if (match) {
|
|
63
|
+
return match
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
export function tryMatchInLexicon(term: CompromiseParsedTerm, lexicon: Lexicon, espeakVoice: string) {
|
|
69
|
+
const shortLanguageCode = getShortLanguageCode(espeakVoice)
|
|
70
|
+
|
|
71
|
+
const lexiconForLanguage = lexicon[shortLanguageCode]
|
|
72
|
+
|
|
73
|
+
if (!lexiconForLanguage) {
|
|
74
|
+
return undefined
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const termText = term.text
|
|
78
|
+
const lowerCaseTermText = termText.toLocaleLowerCase()
|
|
79
|
+
|
|
80
|
+
const entry = lexiconForLanguage[lowerCaseTermText]
|
|
81
|
+
|
|
82
|
+
if (!entry) {
|
|
83
|
+
return undefined
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
for (const substitutionEntry of entry) {
|
|
87
|
+
if (!substitutionEntry.pos || substitutionEntry.pos.includes(term.pos)) {
|
|
88
|
+
const substitutionPhonemesText = substitutionEntry?.pronunciation?.espeak?.[espeakVoice]
|
|
89
|
+
|
|
90
|
+
if (substitutionPhonemesText) {
|
|
91
|
+
const substitutionPhonemes = substitutionPhonemesText.split(/ +/g)
|
|
92
|
+
|
|
93
|
+
return substitutionPhonemes
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return undefined
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export type CompromiseParsedDocument = CompromiseParsedSentence[]
|
|
102
|
+
|
|
103
|
+
export type CompromiseParsedSentence = CompromiseParsedTerm[]
|
|
104
|
+
|
|
105
|
+
export type CompromiseParsedTerm = {
|
|
106
|
+
text: string
|
|
107
|
+
pos: string
|
|
108
|
+
tags: string[]
|
|
109
|
+
preText: string,
|
|
110
|
+
postText: string,
|
|
111
|
+
startOffset: number,
|
|
112
|
+
endOffset: number
|
|
113
|
+
}
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
import * as EspeakTTS from "../synthesis/EspeakTTS.js"
|
|
2
|
+
import { logToStderr } from "../utilities/Utilities.js"
|
|
3
|
+
import * as Segmentation from "./Segmentation.js"
|
|
4
|
+
|
|
5
|
+
const log = logToStderr
|
|
6
|
+
|
|
7
|
+
export async function phonemizeSentence(sentence: string, espeakVoice: string, substitutionMap?: Map<string, string[]>, useIpa = true) {
|
|
8
|
+
const ipaString = await EspeakTTS.textToPhonemes(sentence, espeakVoice, useIpa)
|
|
9
|
+
|
|
10
|
+
const clauseStrings = ipaString.split(" | ")
|
|
11
|
+
|
|
12
|
+
const clauses: string[][][] = []
|
|
13
|
+
|
|
14
|
+
for (let clauseIndex = 0; clauseIndex < clauseStrings.length; clauseIndex++) {
|
|
15
|
+
const clauseString = clauseStrings[clauseIndex]
|
|
16
|
+
|
|
17
|
+
const wordStrings = clauseString.trim().split(/ +/g)
|
|
18
|
+
const words: string[][] = []
|
|
19
|
+
|
|
20
|
+
for (let wordIndex = 0; wordIndex < wordStrings.length; wordIndex++) {
|
|
21
|
+
const word = wordStrings[wordIndex]
|
|
22
|
+
|
|
23
|
+
let wordPhonemes = word.split("_")
|
|
24
|
+
|
|
25
|
+
wordPhonemes = wordPhonemes.flatMap(phoneme => {
|
|
26
|
+
if (!phoneme || phoneme.startsWith("(")) {
|
|
27
|
+
return []
|
|
28
|
+
} else if (phoneme.startsWith("ˈ") || phoneme.startsWith("ˌ")) {
|
|
29
|
+
return [phoneme[0], phoneme.substring(1)]
|
|
30
|
+
} else if (phoneme.endsWith("ˈ") || phoneme.endsWith("ˌ")) {
|
|
31
|
+
return [phoneme.substring(0, phoneme.length - 1), phoneme[phoneme.length - 1]]
|
|
32
|
+
} else {
|
|
33
|
+
return substitutionMap?.get(phoneme) || [phoneme]
|
|
34
|
+
}
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
if (wordPhonemes.length > 0) {
|
|
38
|
+
words.push(wordPhonemes)
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
if (words.length > 0) {
|
|
43
|
+
clauses.push(words)
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
return clauses
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export async function phonemizeText(text: string, voice: string, substitutionMap?: Map<string, string[]>) {
|
|
51
|
+
text = text
|
|
52
|
+
.replaceAll(",", ",")
|
|
53
|
+
.replaceAll("、", ",")
|
|
54
|
+
.replaceAll("。", ".")
|
|
55
|
+
.replaceAll("(", ", ")
|
|
56
|
+
.replaceAll(")", ", ")
|
|
57
|
+
.replaceAll("«", ", ")
|
|
58
|
+
.replaceAll("»", ", ")
|
|
59
|
+
|
|
60
|
+
const segmentedText = await Segmentation.parse(text, voice)
|
|
61
|
+
const preparedClauses: string[] = []
|
|
62
|
+
const clauseBreakers: string[] = []
|
|
63
|
+
|
|
64
|
+
for (const sentence of segmentedText) {
|
|
65
|
+
for (const clause of sentence.phrases) {
|
|
66
|
+
const words = clause.words.filter(wordObject => Segmentation.isWordOrSymbolWord(wordObject.text))
|
|
67
|
+
const preparedClauseText = words.map(word => word.text.replace(/\./g, " ")).join(" ")
|
|
68
|
+
|
|
69
|
+
preparedClauses.push(preparedClauseText)
|
|
70
|
+
|
|
71
|
+
const trimmedClauseText = clause.text.trim()
|
|
72
|
+
const lastChar = trimmedClauseText[trimmedClauseText.length - 1]
|
|
73
|
+
|
|
74
|
+
if (clause.isSentenceFinalizer) {
|
|
75
|
+
if (trimmedClauseText.endsWith('?') || trimmedClauseText.endsWith('?"')) {
|
|
76
|
+
clauseBreakers.push('?')
|
|
77
|
+
} else if (trimmedClauseText.endsWith('!') || trimmedClauseText.endsWith('!"')) {
|
|
78
|
+
clauseBreakers.push('!')
|
|
79
|
+
} else {
|
|
80
|
+
clauseBreakers.push(".")
|
|
81
|
+
}
|
|
82
|
+
} else {
|
|
83
|
+
if (lastChar == ':' || lastChar == ';') {
|
|
84
|
+
clauseBreakers.push(lastChar)
|
|
85
|
+
} else {
|
|
86
|
+
clauseBreakers.push(",")
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
return phonemizeClauses(preparedClauses, voice, clauseBreakers, substitutionMap)
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export async function phonemizeClauses(clauses: string[], voice: string, clauseBreakers: string[], substitutionMap?: Map<string, string[]>) {
|
|
96
|
+
if (clauses.length == 0) {
|
|
97
|
+
return []
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const preparedText = clauses.join("\n\n") // filter(clause => clause.trim().length > 0)
|
|
101
|
+
|
|
102
|
+
const ipaString = await EspeakTTS.textToIPA(preparedText, voice)
|
|
103
|
+
|
|
104
|
+
const ipaLines = ipaString.split("\n")
|
|
105
|
+
|
|
106
|
+
const phonemeLines = ipaLines.map(line => {
|
|
107
|
+
line = line.replace(/_+/g, "_").replace(/ +/g, " ")
|
|
108
|
+
|
|
109
|
+
return line.split(" ").map(word => {
|
|
110
|
+
word = word.replaceAll("_", " ").trim()
|
|
111
|
+
let wordPhonemes = word.split(" ")
|
|
112
|
+
|
|
113
|
+
wordPhonemes = wordPhonemes.flatMap(phoneme => {
|
|
114
|
+
if (!phoneme || phoneme.startsWith("(")) {
|
|
115
|
+
return []
|
|
116
|
+
} else if (phoneme.startsWith("ˈ") || phoneme.startsWith("ˌ")) {
|
|
117
|
+
return [phoneme[0], phoneme.substring(1)]
|
|
118
|
+
} else if (phoneme.endsWith("ˈ") || phoneme.endsWith("ˌ")) {
|
|
119
|
+
return [phoneme.substring(0, phoneme.length - 1), phoneme[phoneme.length - 1]]
|
|
120
|
+
} else {
|
|
121
|
+
return [phoneme]
|
|
122
|
+
}
|
|
123
|
+
})
|
|
124
|
+
|
|
125
|
+
if (substitutionMap) {
|
|
126
|
+
wordPhonemes = wordPhonemes.flatMap(phoneme => substitutionMap.get(phoneme) || [phoneme])
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
return wordPhonemes
|
|
130
|
+
})
|
|
131
|
+
})
|
|
132
|
+
|
|
133
|
+
if (ipaLines.length != clauseBreakers.length) {
|
|
134
|
+
log(clauses)
|
|
135
|
+
log(ipaLines)
|
|
136
|
+
log(clauseBreakers)
|
|
137
|
+
|
|
138
|
+
throw new Error(`Unexpected: IPA lines count (${ipaLines.length}) is not equal to clause breakers count (${clauseBreakers.length})`)
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
for (let i = 0; i < phonemeLines.length; i++) {
|
|
142
|
+
const line = phonemeLines[i]
|
|
143
|
+
const lastWordInLine = line[line.length - 1]
|
|
144
|
+
|
|
145
|
+
lastWordInLine.push(clauseBreakers[i])
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
return phonemeLines
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
export function phonemizedClausesToSentences(phonemizedClauses: string[][][]) {
|
|
152
|
+
let phonemizedSentences: string[][][] = [[]]
|
|
153
|
+
|
|
154
|
+
for (const phonemizedClause of phonemizedClauses) {
|
|
155
|
+
phonemizedSentences[phonemizedSentences.length - 1].push(...phonemizedClause)
|
|
156
|
+
|
|
157
|
+
const lastWord = phonemizedClause[phonemizedClause.length - 1]
|
|
158
|
+
const lastPhoneme = lastWord[lastWord.length - 1]
|
|
159
|
+
|
|
160
|
+
if ([".", "?", "!"].includes(lastPhoneme)) {
|
|
161
|
+
phonemizedSentences.push([])
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
phonemizedSentences = phonemizedSentences.filter(entry => entry.length > 0)
|
|
166
|
+
|
|
167
|
+
return phonemizedSentences
|
|
168
|
+
}
|
package/src/nlp/IPA.ts
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
import { logToStderr } from "../utilities/Utilities.js"
|
|
2
|
+
|
|
3
|
+
const log = logToStderr
|
|
4
|
+
|
|
5
|
+
const consonant = {
|
|
6
|
+
plosive: ["p", "b", "t", "d", "ʈ", "ɖ", "c", "ɟ", "k", "g", "q", "ɢ", "ʔ", /* extensions */ "ɡ"],
|
|
7
|
+
nasal: ["m", "ɱ", "n", "ɳ", "ɲ", "ŋ", "ɴ", "n̩"],
|
|
8
|
+
trill: ["ʙ", "r", "ʀ"],
|
|
9
|
+
tapOrFlap: ["ⱱ", "ɾ", "ɽ"],
|
|
10
|
+
fricative: ["ɸ", "β", "f", "v", "θ", "ð", "s", "z", "ʃ", "ʒ", "ʂ", "ʐ", "ç", "ʝ", "x", "ɣ", "χ", "ʁ", "ħ", "ʕ", "h", "ɦ"],
|
|
11
|
+
lateralFricative: ["ɬ", "ɮ"],
|
|
12
|
+
affricate: ["tʃ", "ʈʃ", "dʒ"], // very incomplete, there are many others
|
|
13
|
+
approximant: ["ʋ", "ɹ", "ɻ", "j", "ɰ", /* extensions */ "w"],
|
|
14
|
+
lateralApproximant: ["l", "ɭ", "ʎ", "ʟ"]
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
const vowel = {
|
|
18
|
+
close: ["i", "yɨ", "ʉɯ", "u", "iː"],
|
|
19
|
+
closeOther: ["ɪ", "ʏ", "ʊ", "ɨ", "ᵻ"],
|
|
20
|
+
closeMid: ["e", "ø", "ɘ", "ɵ", "ɤ", "o", "ə", "oː"],
|
|
21
|
+
openMid: ["ɛ", "œ", "ɜ", "ɞ", "ʌ", "ɔ", "ɜː", "uː", "ɔː", "ɛː"],
|
|
22
|
+
open: ["æ", "a", "ɶ", "ɐ", "ɑ", "ɒ", "ɑː"],
|
|
23
|
+
|
|
24
|
+
rhotic: ["◌˞", "ɚ", "ɝ", "ɹ̩"],
|
|
25
|
+
|
|
26
|
+
diphtongs: [
|
|
27
|
+
"eɪ", "əʊ", "oʊ", "aɪ", "ɔɪ", "aʊ", "iə",
|
|
28
|
+
"ɜr", "ɑr", "ɔr", "oʊr", "oːɹ", "ir", "ɪɹ", "ɔːɹ", "ɑːɹ", "ʊɹ", "ʊr", "ɛr", "ɛɹ",
|
|
29
|
+
"əl", "aɪɚ", "aɪə"
|
|
30
|
+
],
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
let consonants: string[] = []
|
|
34
|
+
|
|
35
|
+
for (const p in consonant) {
|
|
36
|
+
consonants = [...consonants, ...(consonant as any)[p]]
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
let vowels: string[] = []
|
|
40
|
+
|
|
41
|
+
for (const p in vowel) {
|
|
42
|
+
vowels = [...vowels, ...(vowel as any)[p]]
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const all = [" ", ...consonants, ...vowels]
|
|
46
|
+
|
|
47
|
+
export function getPhoneSubstitutionCost1(ipa1: string, ipa2: string) {
|
|
48
|
+
if (ipa1 == ipa2) {
|
|
49
|
+
return 0
|
|
50
|
+
} else {
|
|
51
|
+
return 1
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function getPhoneSubstitutionCost2(ipa1: string, ipa2: string) {
|
|
56
|
+
if (!isKnownSymbol(ipa1)) {
|
|
57
|
+
throw new Error(`'${ipa1}' is not a known IPA symbol`)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
if (!isKnownSymbol(ipa2)) {
|
|
61
|
+
throw new Error(`'${ipa2}' is not a known IPA symbol`)
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
if (ipa1 == ipa2) {
|
|
65
|
+
return 0
|
|
66
|
+
} else if (isVowel(ipa1) && isVowel(ipa2)) {
|
|
67
|
+
return 0.5
|
|
68
|
+
} else if (isConsonant(ipa1) && isConsonant(ipa2)) {
|
|
69
|
+
return 0.5
|
|
70
|
+
} else {
|
|
71
|
+
return 1
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export function getPhoneSubstitutionCost3(ipa1: string, ipa2: string) {
|
|
76
|
+
if (!isKnownSymbol(ipa1)) {
|
|
77
|
+
throw new Error(`'${ipa1}' is not a known IPA symbol`)
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
if (!isKnownSymbol(ipa2)) {
|
|
81
|
+
throw new Error(`'${ipa2}' is not a known IPA symbol`)
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
if (ipa1 == ipa2) {
|
|
85
|
+
return 0
|
|
86
|
+
} else if (isVowel(ipa1) && isVowel(ipa2)) {
|
|
87
|
+
return 0.75
|
|
88
|
+
} else if (isConsonant(ipa1) && isConsonant(ipa2)) {
|
|
89
|
+
if ((isPlosiveOrNasal(ipa1) && isPlosiveOrNasal(ipa2)) ||
|
|
90
|
+
(isTrillTapOrFlap(ipa1) && isTrillTapOrFlap(ipa2)) ||
|
|
91
|
+
(isFricativeLike(ipa1) && isFricativeLike(ipa2)) ||
|
|
92
|
+
(isApproximantLike(ipa1) && isApproximantLike(ipa2))) {
|
|
93
|
+
return 0.5
|
|
94
|
+
} else {
|
|
95
|
+
return 0.75
|
|
96
|
+
}
|
|
97
|
+
} else {
|
|
98
|
+
return 1
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// All symbols:
|
|
103
|
+
export function isKnownSymbol(ipa: string) { return all.includes(ipa) }
|
|
104
|
+
|
|
105
|
+
// All consonants:
|
|
106
|
+
export function isConsonant(ipa: string) { return consonants.includes(ipa) }
|
|
107
|
+
|
|
108
|
+
// All vowels:
|
|
109
|
+
export function isVowel(ipa: string) { return vowels.includes(ipa) }
|
|
110
|
+
|
|
111
|
+
// Higher level grouping:
|
|
112
|
+
export function isPlosiveOrNasal(ipa: string) { return isPlosive(ipa) || isNasal(ipa) }
|
|
113
|
+
|
|
114
|
+
export function isTrillTapOrFlap(ipa: string) { return isTrill(ipa) || isTapOrFlap(ipa) }
|
|
115
|
+
|
|
116
|
+
export function isFricativeLike(ipa: string) { return isFricative(ipa) || isLateralFricative(ipa) || isAffricate(ipa) }
|
|
117
|
+
|
|
118
|
+
export function isApproximantLike(ipa: string) { return isApproximant(ipa) || isLateralApproximant(ipa) }
|
|
119
|
+
|
|
120
|
+
// Grouping:
|
|
121
|
+
export function isPlosive(ipa: string) { return consonant.plosive.includes(ipa) }
|
|
122
|
+
|
|
123
|
+
export function isNasal(ipa: string) { return consonant.nasal.includes(ipa) }
|
|
124
|
+
|
|
125
|
+
export function isTrill(ipa: string) { return consonant.trill.includes(ipa) }
|
|
126
|
+
|
|
127
|
+
export function isTapOrFlap(ipa: string) { return consonant.tapOrFlap.includes(ipa) }
|
|
128
|
+
|
|
129
|
+
export function isFricative(ipa: string) { return consonant.fricative.includes(ipa) }
|
|
130
|
+
|
|
131
|
+
export function isLateralFricative(ipa: string) { return consonant.lateralFricative.includes(ipa) }
|
|
132
|
+
|
|
133
|
+
export function isAffricate(ipa: string) { return consonant.affricate.includes(ipa) }
|
|
134
|
+
|
|
135
|
+
export function isApproximant(ipa: string) { return consonant.approximant.includes(ipa) }
|
|
136
|
+
|
|
137
|
+
export function isLateralApproximant(ipa: string) { return consonant.lateralApproximant.includes(ipa) }
|
|
138
|
+
|
|
139
|
+
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import path from "path"
|
|
2
|
+
import { OpenPromise } from "../utilities/OpenPromise.js"
|
|
3
|
+
import { resolveModuleScriptPath } from "../utilities/Utilities.js"
|
|
4
|
+
|
|
5
|
+
export async function splitJapaneseTextToWords_Kuromoji(text: string) {
|
|
6
|
+
const tokenizer = await getKuromojiTokenizer()
|
|
7
|
+
|
|
8
|
+
const results: any[] = tokenizer.tokenize(text)
|
|
9
|
+
const words = results.map(entry => entry.surface_form)
|
|
10
|
+
|
|
11
|
+
return words
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
let kuromojiTokenizer: any
|
|
15
|
+
|
|
16
|
+
async function getKuromojiTokenizer() {
|
|
17
|
+
if (kuromojiTokenizer) {
|
|
18
|
+
return kuromojiTokenizer
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const { default: kuromoji } = await import("kuromoji")
|
|
22
|
+
|
|
23
|
+
const resultOpenPromise = new OpenPromise<any>()
|
|
24
|
+
|
|
25
|
+
const kuromojiScriptPath = await resolveModuleScriptPath('kuromoji')
|
|
26
|
+
const dictionaryPath = path.join(path.dirname(kuromojiScriptPath), "..", '/dict')
|
|
27
|
+
|
|
28
|
+
kuromoji.builder({ dicPath: dictionaryPath }).build(function (error: any, tokenizer: any) {
|
|
29
|
+
if (error) {
|
|
30
|
+
resultOpenPromise.reject(error)
|
|
31
|
+
return
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
kuromojiTokenizer = tokenizer
|
|
35
|
+
|
|
36
|
+
resultOpenPromise.resolve(kuromojiTokenizer)
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
return resultOpenPromise.promise
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/*
|
|
43
|
+
export async function splitJapaneseTextToWords_Sudachi(text: string, mode: 0 | 1 | 2) {
|
|
44
|
+
const { TokenizeMode, tokenize } = await import("sudachi")
|
|
45
|
+
|
|
46
|
+
const resultString = tokenize(text, mode)
|
|
47
|
+
|
|
48
|
+
const parsedResult: any[] = JSON.parse(resultString)
|
|
49
|
+
const result = parsedResult.map(entry => entry.surface)
|
|
50
|
+
|
|
51
|
+
return result
|
|
52
|
+
}
|
|
53
|
+
*/
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { readAndParseJsonFile, resolveToModuleRootDir } from "../utilities/FileSystem.js"
|
|
2
|
+
import { getShortLanguageCode } from "../utilities/Locale.js"
|
|
3
|
+
|
|
4
|
+
export function tryGetFirstLexiconSubstitution(sentenceWords: string[], wordIndex: number, lexicons: Lexicon[], languageCode: string) {
|
|
5
|
+
const reversedLexicons = [...lexicons].reverse() // Give precedence to later lexicons
|
|
6
|
+
|
|
7
|
+
for (const lexicon of reversedLexicons) {
|
|
8
|
+
const match = tryGetLexiconSubstitution(sentenceWords, wordIndex, lexicon, languageCode)
|
|
9
|
+
|
|
10
|
+
if (match) {
|
|
11
|
+
return match
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
return undefined
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export function tryGetLexiconSubstitution(sentenceWords: string[], wordIndex: number, lexicon: Lexicon, languageCode: string) {
|
|
19
|
+
let word = sentenceWords[wordIndex]
|
|
20
|
+
|
|
21
|
+
if (!word) {
|
|
22
|
+
return
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const shortLanguageCode = getShortLanguageCode(languageCode)
|
|
26
|
+
const lexiconForLanguage = lexicon[shortLanguageCode]
|
|
27
|
+
|
|
28
|
+
if (!lexiconForLanguage) {
|
|
29
|
+
return
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
const lexiconEntry = lexiconForLanguage[word]
|
|
33
|
+
|
|
34
|
+
if (!lexiconEntry) {
|
|
35
|
+
return
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
for (let i = 0; i < lexiconEntry.length; i++) {
|
|
39
|
+
const substitutionEntry = lexiconEntry[i]
|
|
40
|
+
|
|
41
|
+
const substitutionPhonemesText = substitutionEntry?.pronunciation?.espeak?.[languageCode]
|
|
42
|
+
|
|
43
|
+
if (!substitutionPhonemesText) {
|
|
44
|
+
continue
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
const precedingWord = sentenceWords[wordIndex - 1] || ""
|
|
48
|
+
const succeedingWord = sentenceWords[wordIndex + 1] || ""
|
|
49
|
+
|
|
50
|
+
const precededBy = substitutionEntry?.precededBy || []
|
|
51
|
+
const notPrecededBy = substitutionEntry?.notPrecededBy || []
|
|
52
|
+
|
|
53
|
+
const succeededBy = substitutionEntry?.succeededBy || []
|
|
54
|
+
const notSucceededBy = substitutionEntry?.notSucceededBy || []
|
|
55
|
+
|
|
56
|
+
const hasNegativePattern = notPrecededBy.includes(precedingWord) || notSucceededBy.includes(succeedingWord)
|
|
57
|
+
const hasPositivePattern = precededBy.includes(precedingWord) || succeededBy.includes(succeedingWord)
|
|
58
|
+
|
|
59
|
+
if (i == lexiconEntry.length - 1 || (hasPositivePattern && !hasNegativePattern)) {
|
|
60
|
+
const substitutionPhonemes = substitutionPhonemesText.split(/ +/g)
|
|
61
|
+
|
|
62
|
+
return substitutionPhonemes
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export async function loadLexiconFile(jsonFilePath: string): Promise<Lexicon> {
|
|
68
|
+
const parsedLexicon: Lexicon = await readAndParseJsonFile(jsonFilePath)
|
|
69
|
+
|
|
70
|
+
return parsedLexicon
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export async function loadLexiconsForLanguage(language: string, customLexiconPaths?: string[]) {
|
|
74
|
+
const lexicons: Lexicon[] = []
|
|
75
|
+
|
|
76
|
+
if (getShortLanguageCode(language) == "en") {
|
|
77
|
+
const heteronymsLexicon = await loadLexiconFile(resolveToModuleRootDir("data/lexicons/heteronyms.en.json"))
|
|
78
|
+
lexicons.push(heteronymsLexicon)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
if (customLexiconPaths && customLexiconPaths.length > 0) {
|
|
82
|
+
for (const customLexicon of customLexiconPaths) {
|
|
83
|
+
const customLexiconObject = await loadLexiconFile(customLexicon)
|
|
84
|
+
|
|
85
|
+
lexicons.push(customLexiconObject)
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
return lexicons
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export type Lexicon = {
|
|
93
|
+
[shortLanguageCode: string]: LexiconForLanguage
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export type LexiconForLanguage = {
|
|
97
|
+
[word: string]: LexiconEntry[]
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export type LexiconEntry = {
|
|
101
|
+
pos?: string[]
|
|
102
|
+
case?: LexiconWordCase
|
|
103
|
+
|
|
104
|
+
pronunciation?: {
|
|
105
|
+
espeak?: LexiconPronunciationForLanguageCodes
|
|
106
|
+
sapi?: LexiconPronunciationForLanguageCodes
|
|
107
|
+
},
|
|
108
|
+
|
|
109
|
+
precededBy?: string[]
|
|
110
|
+
notPrecededBy?: string[]
|
|
111
|
+
|
|
112
|
+
succeededBy?: string[]
|
|
113
|
+
notSucceededBy?: string[]
|
|
114
|
+
|
|
115
|
+
example?: string
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export type LexiconWordCase = "any" | "capitalized" | "uppercase" | "lowercase" | "titlecase" | "camelcase" | "pascalcase"
|
|
119
|
+
export type LexiconPronunciationForLanguageCodes = { [languageCode: string]: string }
|