use-voice-control 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/core/deepgram.d.ts +5 -0
- package/dist/core/kokoro.d.ts +5 -0
- package/dist/index.d.ts +25 -0
- package/dist/index.js +102 -0
- package/dist/index.js.map +1 -0
- package/dist/types/types.d.ts +17 -0
- package/package.json +7 -11
- package/speech/core/KokoroTTS.js +91 -0
- package/speech/core/deepgram.ts +56 -0
- package/speech/core/kokoro.js +93 -0
- package/speech/core/kokoro.ts +81 -0
- package/speech/docs/ARCHITECTURE.md +349 -0
- package/speech/docs/HUGGINGFACE_MIGRATION.md +172 -0
- package/speech/docs/INTEGRATION.md +276 -0
- package/speech/docs/MIGRATION.md +220 -0
- package/speech/docs/QUICKSTART.md +159 -0
- package/speech/docs/README.md +167 -0
- package/speech/index.ts +54 -0
- package/speech/legacy/conversation.js +150 -0
- package/speech/legacy/main.js +56 -0
- package/speech/legacy/stt.js +161 -0
- package/speech/legacy/tts.js +56 -0
- package/speech/legacy/worker.js +71 -0
- package/speech/types/types.ts +34 -0
- package/speech/ui/AudioPlayer.js +84 -0
- package/speech/ui/ui.js +109 -0
- package/speech/ui/voice-selector.js +77 -0
- package/speech/ui/voices.js +259 -0
- package/speech/utils/audio-utils.js +82 -0
- package/speech/utils/phonemize.js +198 -0
- package/speech/utils/semantic-split.js +107 -0
- package/speech/utils/sentence-detector.js +89 -0
- package/readme.md +0 -1
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
//see also: https://github.com/colmeye/js-mediarecorder-to-wav/blob/main/wave-worker.js
|
|
2
|
+
|
|
3
|
+
// 16000
|
|
4
|
+
export async function resampleAudio(audioBuffer, targetSampleRate) {
|
|
5
|
+
const offlineCtx = new OfflineAudioContext(1, targetSampleRate * audioBuffer.duration, targetSampleRate);
|
|
6
|
+
const source = offlineCtx.createBufferSource();
|
|
7
|
+
source.buffer = audioBuffer;
|
|
8
|
+
source.connect(offlineCtx.destination);
|
|
9
|
+
source.start(0);
|
|
10
|
+
return await offlineCtx.startRendering();
|
|
11
|
+
}
|
|
12
|
+
// const floatArray = resampledBuffer.getChannelData(0);
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Applies gain to an audio buffer to increase or decrease volume
|
|
16
|
+
* @param {AudioBuffer} audioBuffer - The audio buffer to adjust
|
|
17
|
+
* @param {number} gain - Gain factor (e.g., 1.5 for 50% louder)
|
|
18
|
+
* @returns {AudioBuffer} - The adjusted audio buffer
|
|
19
|
+
*/
|
|
20
|
+
export function applyAudioGain(audioBuffer, gain = 1.5) {
|
|
21
|
+
// Create a copy to avoid modifying the original
|
|
22
|
+
const audioContext = new AudioContext();
|
|
23
|
+
const newBuffer = audioContext.createBuffer(
|
|
24
|
+
audioBuffer.numberOfChannels,
|
|
25
|
+
audioBuffer.length,
|
|
26
|
+
audioBuffer.sampleRate
|
|
27
|
+
);
|
|
28
|
+
|
|
29
|
+
// Apply gain to each channel
|
|
30
|
+
for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
|
|
31
|
+
const channelData = audioBuffer.getChannelData(i);
|
|
32
|
+
const newChannelData = newBuffer.getChannelData(i);
|
|
33
|
+
for (let j = 0; j < channelData.length; j++) {
|
|
34
|
+
// Apply gain and clamp to [-1, 1]
|
|
35
|
+
newChannelData[j] = Math.max(-1, Math.min(1, channelData[j] * gain));
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
return newBuffer;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export function convertAudioBufferToWav(audioBuffer) {
|
|
43
|
+
const numOfChannels = audioBuffer.numberOfChannels;
|
|
44
|
+
const length = audioBuffer.length * numOfChannels * 2; // 2 bytes per sample
|
|
45
|
+
const sampleRate = audioBuffer.sampleRate;
|
|
46
|
+
const buffer = new ArrayBuffer(44 + length);
|
|
47
|
+
const view = new DataView(buffer);
|
|
48
|
+
|
|
49
|
+
writeString(view, 0, 'RIFF');
|
|
50
|
+
view.setUint32(4, 36 + length, true);
|
|
51
|
+
writeString(view, 8, 'WAVE');
|
|
52
|
+
|
|
53
|
+
writeString(view, 12, 'fmt ');
|
|
54
|
+
view.setUint32(16, 16, true); // subchunk1 size
|
|
55
|
+
view.setUint16(20, 1, true); // PCM format
|
|
56
|
+
view.setUint16(22, numOfChannels, true);
|
|
57
|
+
view.setUint32(24, sampleRate, true);
|
|
58
|
+
view.setUint32(28, sampleRate * numOfChannels * 2, true); // byte rate
|
|
59
|
+
view.setUint16(32, numOfChannels * 2, true); // block align
|
|
60
|
+
view.setUint16(34, 16, true); // bits per sample
|
|
61
|
+
|
|
62
|
+
writeString(view, 36, 'data');
|
|
63
|
+
view.setUint32(40, length, true);
|
|
64
|
+
|
|
65
|
+
let offset = 44;
|
|
66
|
+
for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
|
|
67
|
+
const channelData = audioBuffer.getChannelData(i);
|
|
68
|
+
for (let j = 0; j < channelData.length; j++) {
|
|
69
|
+
const sample = Math.max(-1, Math.min(1, channelData[j]));
|
|
70
|
+
view.setInt16(offset, sample * 0x7FFF, true);
|
|
71
|
+
offset += 2;
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
return buffer;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function writeString(view, offset, string) {
|
|
79
|
+
for (let i = 0; i < string.length; i++) {
|
|
80
|
+
view.setUint8(offset + i, string.charCodeAt(i));
|
|
81
|
+
}
|
|
82
|
+
}
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
import { phonemize as espeakng } from "https://cdn.jsdelivr.net/npm/phonemizer@1.2.1/dist/phonemizer.min.js";
|
|
2
|
+
//import { phonemize as espeakng } from "phonemizer";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Helper function to split a string on a regex, but keep the delimiters.
|
|
6
|
+
* This is required, because the JavaScript `.split()` method does not keep the delimiters,
|
|
7
|
+
* and wrapping in a capturing group causes issues with existing capturing groups (due to nesting).
|
|
8
|
+
* @param {string} text The text to split.
|
|
9
|
+
* @param {RegExp} regex The regex to split on.
|
|
10
|
+
* @returns {{match: boolean; text: string}[]} The split string.
|
|
11
|
+
*/
|
|
12
|
+
function split(text, regex) {
|
|
13
|
+
const result = [];
|
|
14
|
+
let prev = 0;
|
|
15
|
+
for (const match of text.matchAll(regex)) {
|
|
16
|
+
const fullMatch = match[0];
|
|
17
|
+
if (prev < match.index) {
|
|
18
|
+
result.push({ match: false, text: text.slice(prev, match.index) });
|
|
19
|
+
}
|
|
20
|
+
if (fullMatch.length > 0) {
|
|
21
|
+
result.push({ match: true, text: fullMatch });
|
|
22
|
+
}
|
|
23
|
+
prev = match.index + fullMatch.length;
|
|
24
|
+
}
|
|
25
|
+
if (prev < text.length) {
|
|
26
|
+
result.push({ match: false, text: text.slice(prev) });
|
|
27
|
+
}
|
|
28
|
+
return result;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Helper function to split numbers into phonetic equivalents
|
|
33
|
+
* @param {string} match The matched number
|
|
34
|
+
* @returns {string} The phonetic equivalent
|
|
35
|
+
*/
|
|
36
|
+
function split_num(match) {
|
|
37
|
+
if (match.includes(".")) {
|
|
38
|
+
return match;
|
|
39
|
+
} else if (match.includes(":")) {
|
|
40
|
+
let [h, m] = match.split(":").map(Number);
|
|
41
|
+
if (m === 0) {
|
|
42
|
+
return `${h} o'clock`;
|
|
43
|
+
} else if (m < 10) {
|
|
44
|
+
return `${h} oh ${m}`;
|
|
45
|
+
}
|
|
46
|
+
return `${h} ${m}`;
|
|
47
|
+
}
|
|
48
|
+
let year = parseInt(match.slice(0, 4), 10);
|
|
49
|
+
if (year < 1100 || year % 1000 < 10) {
|
|
50
|
+
return match;
|
|
51
|
+
}
|
|
52
|
+
let left = match.slice(0, 2);
|
|
53
|
+
let right = parseInt(match.slice(2, 4), 10);
|
|
54
|
+
let suffix = match.endsWith("s") ? "s" : "";
|
|
55
|
+
if (year % 1000 >= 100 && year % 1000 <= 999) {
|
|
56
|
+
if (right === 0) {
|
|
57
|
+
return `${left} hundred${suffix}`;
|
|
58
|
+
} else if (right < 10) {
|
|
59
|
+
return `${left} oh ${right}${suffix}`;
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
return `${left} ${right}${suffix}`;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Helper function to format monetary values
|
|
67
|
+
* @param {string} match The matched currency
|
|
68
|
+
* @returns {string} The formatted currency
|
|
69
|
+
*/
|
|
70
|
+
function flip_money(match) {
|
|
71
|
+
const bill = match[0] === "$" ? "dollar" : "pound";
|
|
72
|
+
if (isNaN(Number(match.slice(1)))) {
|
|
73
|
+
return `${match.slice(1)} ${bill}s`;
|
|
74
|
+
} else if (!match.includes(".")) {
|
|
75
|
+
let suffix = match.slice(1) === "1" ? "" : "s";
|
|
76
|
+
return `${match.slice(1)} ${bill}${suffix}`;
|
|
77
|
+
}
|
|
78
|
+
const [b, c] = match.slice(1).split(".");
|
|
79
|
+
const d = parseInt(c.padEnd(2, "0"), 10);
|
|
80
|
+
let coins = match[0] === "$" ? (d === 1 ? "cent" : "cents") : d === 1 ? "penny" : "pence";
|
|
81
|
+
return `${b} ${bill}${b === "1" ? "" : "s"} and ${d} ${coins}`;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Helper function to process decimal numbers
|
|
86
|
+
* @param {string} match The matched number
|
|
87
|
+
* @returns {string} The formatted number
|
|
88
|
+
*/
|
|
89
|
+
function point_num(match) {
|
|
90
|
+
let [a, b] = match.split(".");
|
|
91
|
+
return `${a} point ${b.split("").join(" ")}`;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Normalize text for phonemization
|
|
96
|
+
* @param {string} text The text to normalize
|
|
97
|
+
* @returns {string} The normalized text
|
|
98
|
+
*/
|
|
99
|
+
function normalize_text(text) {
|
|
100
|
+
return (
|
|
101
|
+
text
|
|
102
|
+
// 1. Handle quotes and brackets
|
|
103
|
+
.replace(/[‘’]/g, "'")
|
|
104
|
+
.replace(/«/g, "“")
|
|
105
|
+
.replace(/»/g, "”")
|
|
106
|
+
.replace(/[“”]/g, '"')
|
|
107
|
+
.replace(/\(/g, "«")
|
|
108
|
+
.replace(/\)/g, "»")
|
|
109
|
+
|
|
110
|
+
// 2. Replace uncommon punctuation marks
|
|
111
|
+
.replace(/、/g, ", ")
|
|
112
|
+
.replace(/。/g, ". ")
|
|
113
|
+
.replace(/!/g, "! ")
|
|
114
|
+
.replace(/,/g, ", ")
|
|
115
|
+
.replace(/:/g, ": ")
|
|
116
|
+
.replace(/;/g, "; ")
|
|
117
|
+
.replace(/?/g, "? ")
|
|
118
|
+
|
|
119
|
+
// 3. Whitespace normalization
|
|
120
|
+
.replace(/[^\S \n]/g, " ")
|
|
121
|
+
.replace(/ +/, " ")
|
|
122
|
+
.replace(/(?<=\n) +(?=\n)/g, "")
|
|
123
|
+
|
|
124
|
+
// 4. Abbreviations
|
|
125
|
+
.replace(/\bD[Rr]\.(?= [A-Z])/g, "Doctor")
|
|
126
|
+
.replace(/\b(?:Mr\.|MR\.(?= [A-Z]))/g, "Mister")
|
|
127
|
+
.replace(/\b(?:Ms\.|MS\.(?= [A-Z]))/g, "Miss")
|
|
128
|
+
.replace(/\b(?:Mrs\.|MRS\.(?= [A-Z]))/g, "Mrs")
|
|
129
|
+
.replace(/\betc\.(?! [A-Z])/gi, "etc")
|
|
130
|
+
|
|
131
|
+
// 5. Normalize casual words
|
|
132
|
+
.replace(/\b(y)eah?\b/gi, "$1e'a")
|
|
133
|
+
|
|
134
|
+
// 5. Handle numbers and currencies
|
|
135
|
+
.replace(/\d*\.\d+|\b\d{4}s?\b|(?<!:)\b(?:[1-9]|1[0-2]):[0-5]\d\b(?!:)/g, split_num)
|
|
136
|
+
.replace(/(?<=\d),(?=\d)/g, "")
|
|
137
|
+
.replace(/[$£]\d+(?:\.\d+)?(?: hundred| thousand| (?:[bm]|tr)illion)*\b|[$£]\d+\.\d\d?\b/gi, flip_money)
|
|
138
|
+
.replace(/\d*\.\d+/g, point_num)
|
|
139
|
+
.replace(/(?<=\d)-(?=\d)/g, " to ")
|
|
140
|
+
.replace(/(?<=\d)S/g, " S")
|
|
141
|
+
|
|
142
|
+
// 6. Handle possessives
|
|
143
|
+
.replace(/(?<=[BCDFGHJ-NP-TV-Z])'?s\b/g, "'S")
|
|
144
|
+
.replace(/(?<=X')S\b/g, "s")
|
|
145
|
+
|
|
146
|
+
// 7. Handle hyphenated words/letters
|
|
147
|
+
.replace(/(?:[A-Za-z]\.){2,} [a-z]/g, (m) => m.replace(/\./g, "-"))
|
|
148
|
+
.replace(/(?<=[A-Z])\.(?=[A-Z])/gi, "-")
|
|
149
|
+
|
|
150
|
+
// 8. Strip leading and trailing whitespace
|
|
151
|
+
.trim()
|
|
152
|
+
);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Escapes regular expression special characters from a string by replacing them with their escaped counterparts.
|
|
157
|
+
*
|
|
158
|
+
* @param {string} string The string to escape.
|
|
159
|
+
* @returns {string} The escaped string.
|
|
160
|
+
*/
|
|
161
|
+
function escapeRegExp(string) {
|
|
162
|
+
return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); // $& means the whole matched string
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
const PUNCTUATION = ';:,.!?¡¿—…"«»“”(){}[]';
|
|
166
|
+
const PUNCTUATION_PATTERN = new RegExp(`(\\s*[${escapeRegExp(PUNCTUATION)}]+\\s*)+`, "g");
|
|
167
|
+
|
|
168
|
+
export async function phonemize(text, language = "a", norm = true) {
|
|
169
|
+
// 1. Normalize text
|
|
170
|
+
if (norm) {
|
|
171
|
+
text = normalize_text(text);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// 2. Split into chunks, to ensure we preserve punctuation
|
|
175
|
+
const sections = split(text, PUNCTUATION_PATTERN);
|
|
176
|
+
|
|
177
|
+
// 3. Convert each section to phonemes
|
|
178
|
+
const lang = language === "a" ? "en-us" : "en";
|
|
179
|
+
const ps = (await Promise.all(sections.map(async ({ match, text }) => (match ? text : (await espeakng(text, lang)).join(" "))))).join("");
|
|
180
|
+
|
|
181
|
+
// 4. Post-process phonemes
|
|
182
|
+
let processed = ps
|
|
183
|
+
// https://en.wiktionary.org/wiki/kokoro#English
|
|
184
|
+
.replace(/kəkˈoːɹoʊ/g, "kˈoʊkəɹoʊ")
|
|
185
|
+
.replace(/kəkˈɔːɹəʊ/g, "kˈəʊkəɹəʊ")
|
|
186
|
+
.replace(/ʲ/g, "j")
|
|
187
|
+
.replace(/r/g, "ɹ")
|
|
188
|
+
.replace(/x/g, "k")
|
|
189
|
+
.replace(/ɬ/g, "l")
|
|
190
|
+
.replace(/(?<=[a-zɹː])(?=hˈʌndɹɪd)/g, " ")
|
|
191
|
+
.replace(/ z(?=[;:,.!?¡¿—…"«»“” ]|$)/g, "z");
|
|
192
|
+
|
|
193
|
+
// 5. Additional post-processing for American English
|
|
194
|
+
if (language === "a") {
|
|
195
|
+
processed = processed.replace(/(?<=nˈaɪn)ti(?!ː)/g, "di");
|
|
196
|
+
}
|
|
197
|
+
return processed.trim();
|
|
198
|
+
}
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
export function splitTextSmart(text, maxChunkLength = 500) {
|
|
2
|
+
const paragraphChunks = text.split(/\n\s*\n/);
|
|
3
|
+
const finalChunks = [];
|
|
4
|
+
|
|
5
|
+
for (let para of paragraphChunks) {
|
|
6
|
+
if (para.length <= maxChunkLength) {
|
|
7
|
+
finalChunks.push(para.trim());
|
|
8
|
+
continue;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
const sentenceRegex = /(?<=[.?!])(?=\s+["“”'a-z])/gi;
|
|
12
|
+
const sentences = para.split(sentenceRegex);
|
|
13
|
+
|
|
14
|
+
let chunk = '';
|
|
15
|
+
for (let sentence of sentences) {
|
|
16
|
+
sentence = sentence.trim();
|
|
17
|
+
|
|
18
|
+
if (sentence.length > maxChunkLength) {
|
|
19
|
+
// Sentence too long — fallback split
|
|
20
|
+
const subChunks = splitLongSentence(sentence, maxChunkLength);
|
|
21
|
+
for (let sub of subChunks) {
|
|
22
|
+
if ((chunk + ' ' + sub).length > maxChunkLength) {
|
|
23
|
+
if (chunk) finalChunks.push(chunk.trim());
|
|
24
|
+
chunk = sub;
|
|
25
|
+
} else {
|
|
26
|
+
chunk += (chunk ? ' ' : '') + sub;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
continue;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
if ((chunk + ' ' + sentence).length > maxChunkLength) {
|
|
33
|
+
if (chunk) finalChunks.push(chunk.trim());
|
|
34
|
+
chunk = sentence;
|
|
35
|
+
} else {
|
|
36
|
+
chunk += (chunk ? ' ' : '') + sentence;
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
if (chunk) finalChunks.push(chunk.trim());
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
return finalChunks;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function splitLongSentence(sentence, maxLen) {
|
|
46
|
+
const chunks = [];
|
|
47
|
+
let current = '';
|
|
48
|
+
|
|
49
|
+
const commaParts = sentence.split(/,\s*/);
|
|
50
|
+
for (let part of commaParts) {
|
|
51
|
+
if ((current + ', ' + part).length > maxLen) {
|
|
52
|
+
if (current) chunks.push(current.trim());
|
|
53
|
+
if (part.length > maxLen) {
|
|
54
|
+
const words = part.split(/\s+/);
|
|
55
|
+
let wordChunk = '';
|
|
56
|
+
for (let word of words) {
|
|
57
|
+
if ((wordChunk + ' ' + word).length > maxLen) {
|
|
58
|
+
if (wordChunk) chunks.push(wordChunk.trim());
|
|
59
|
+
wordChunk = word;
|
|
60
|
+
} else {
|
|
61
|
+
wordChunk += (wordChunk ? ' ' : '') + word;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
if (wordChunk) chunks.push(wordChunk.trim());
|
|
65
|
+
current = '';
|
|
66
|
+
} else {
|
|
67
|
+
current = part;
|
|
68
|
+
}
|
|
69
|
+
} else {
|
|
70
|
+
current += (current ? ', ' : '') + part;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
if (current) chunks.push(current.trim());
|
|
74
|
+
|
|
75
|
+
return chunks;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
function splitTextSmartOld(text, maxChunkLength = 500) {
|
|
80
|
+
const paragraphChunks = text.split(/\n\s*\n/); // Step 1: split on double returns
|
|
81
|
+
const finalChunks = [];
|
|
82
|
+
|
|
83
|
+
for (let para of paragraphChunks) {
|
|
84
|
+
if (para.length <= maxChunkLength) {
|
|
85
|
+
finalChunks.push(para.trim());
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Step 2: Further split on sentence boundaries if too long
|
|
90
|
+
const sentenceRegex = /(?<=[.?!])(?=\s+["“”'a-z])/gi;
|
|
91
|
+
const sentences = para.split(sentenceRegex);
|
|
92
|
+
|
|
93
|
+
let chunk = '';
|
|
94
|
+
for (let sentence of sentences) {
|
|
95
|
+
sentence = sentence.trim();
|
|
96
|
+
if ((chunk + ' ' + sentence).length > maxChunkLength) {
|
|
97
|
+
if (chunk) finalChunks.push(chunk.trim());
|
|
98
|
+
chunk = sentence;
|
|
99
|
+
} else {
|
|
100
|
+
chunk += (chunk ? ' ' : '') + sentence;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
if (chunk) finalChunks.push(chunk.trim());
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
return finalChunks;
|
|
107
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
|
|
2
|
+
/**
|
|
3
|
+
* Utility functions for detecting sentence boundaries in streaming text
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
// These are common sentence ending patterns
|
|
7
|
+
const SENTENCE_PATTERNS = {
|
|
8
|
+
// Basic sentence endings: period, exclamation, question mark followed by space or end
|
|
9
|
+
basicEnd: /[.!?][\s"')\]]*($|(?=\s*[A-Z]))/,
|
|
10
|
+
|
|
11
|
+
// Dialog endings: quote followed by punctuation and space or end
|
|
12
|
+
dialogEnd: /["'][,.!?][\s"')\]]*($|(?=\s*[A-Z]))/,
|
|
13
|
+
|
|
14
|
+
// List item or enumeration endings: semicolon, colon
|
|
15
|
+
listItemEnd: /[;:][\s]*($|(?=\s*[-•*]))/,
|
|
16
|
+
|
|
17
|
+
// Paragraph breaks: double newline
|
|
18
|
+
paragraphBreak: /\n\s*\n/,
|
|
19
|
+
|
|
20
|
+
// Force breaks for very long text without clear sentence boundaries
|
|
21
|
+
longText: (text) => text.length > 150
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Determines if a text contains a complete sentence or should be sent for TTS
|
|
26
|
+
* @param {string} text - The text to check
|
|
27
|
+
* @param {Object} options - Configuration options
|
|
28
|
+
* @returns {boolean} - True if the text contains a complete sentence or should be spoken
|
|
29
|
+
*/
|
|
30
|
+
export function isCompleteSentence(text) {
|
|
31
|
+
if (!text || text.trim().length === 0) {
|
|
32
|
+
return false;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// Check all patterns
|
|
36
|
+
return (
|
|
37
|
+
SENTENCE_PATTERNS.basicEnd.test(text) ||
|
|
38
|
+
SENTENCE_PATTERNS.dialogEnd.test(text) ||
|
|
39
|
+
SENTENCE_PATTERNS.listItemEnd.test(text) ||
|
|
40
|
+
SENTENCE_PATTERNS.paragraphBreak.test(text) ||
|
|
41
|
+
SENTENCE_PATTERNS.longText(text)
|
|
42
|
+
);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Process streaming text and return complete sentences
|
|
47
|
+
*
|
|
48
|
+
* @param {string} accumulator - The text accumulated so far
|
|
49
|
+
* @param {string} newContent - The new content to add
|
|
50
|
+
* @returns {Object} - Object with processed results:
|
|
51
|
+
* - sentences: Array of complete sentences to speak
|
|
52
|
+
* - remainder: Remaining text that doesn't form a complete sentence yet
|
|
53
|
+
*/
|
|
54
|
+
export function processStreamingText(accumulator, newContent) {
|
|
55
|
+
const combinedText = accumulator + newContent;
|
|
56
|
+
|
|
57
|
+
// If the text is empty, just return
|
|
58
|
+
if (!combinedText || combinedText.trim().length === 0) {
|
|
59
|
+
return { sentences: [], remainder: '' };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Split text by common sentence boundaries to check for multiple sentences
|
|
63
|
+
const parts = combinedText.split(/(?<=[.!?][\s"')\]])/);
|
|
64
|
+
|
|
65
|
+
// If we only have one part and it's not a complete sentence
|
|
66
|
+
if (parts.length === 1 && !isCompleteSentence(parts[0])) {
|
|
67
|
+
return { sentences: [], remainder: combinedText };
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Otherwise process all parts
|
|
71
|
+
const sentences = [];
|
|
72
|
+
let currentSentence = '';
|
|
73
|
+
|
|
74
|
+
for (let i = 0; i < parts.length; i++) {
|
|
75
|
+
currentSentence += parts[i];
|
|
76
|
+
|
|
77
|
+
// If this forms a complete sentence or is the last part that should be spoken
|
|
78
|
+
if (isCompleteSentence(currentSentence) ||
|
|
79
|
+
(i === parts.length - 1 && SENTENCE_PATTERNS.longText(currentSentence))) {
|
|
80
|
+
sentences.push(currentSentence.trim());
|
|
81
|
+
currentSentence = '';
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
return {
|
|
86
|
+
sentences,
|
|
87
|
+
remainder: currentSentence
|
|
88
|
+
};
|
|
89
|
+
}
|
package/readme.md
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-

|