use-voice-control 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,82 @@
1
+ //see also: https://github.com/colmeye/js-mediarecorder-to-wav/blob/main/wave-worker.js
2
+
3
+ // 16000
4
+ export async function resampleAudio(audioBuffer, targetSampleRate) {
5
+ const offlineCtx = new OfflineAudioContext(1, targetSampleRate * audioBuffer.duration, targetSampleRate);
6
+ const source = offlineCtx.createBufferSource();
7
+ source.buffer = audioBuffer;
8
+ source.connect(offlineCtx.destination);
9
+ source.start(0);
10
+ return await offlineCtx.startRendering();
11
+ }
12
+ // const floatArray = resampledBuffer.getChannelData(0);
13
+
14
+ /**
15
+ * Applies gain to an audio buffer to increase or decrease volume
16
+ * @param {AudioBuffer} audioBuffer - The audio buffer to adjust
17
+ * @param {number} gain - Gain factor (e.g., 1.5 for 50% louder)
18
+ * @returns {AudioBuffer} - The adjusted audio buffer
19
+ */
20
+ export function applyAudioGain(audioBuffer, gain = 1.5) {
21
+ // Create a copy to avoid modifying the original
22
+ const audioContext = new AudioContext();
23
+ const newBuffer = audioContext.createBuffer(
24
+ audioBuffer.numberOfChannels,
25
+ audioBuffer.length,
26
+ audioBuffer.sampleRate
27
+ );
28
+
29
+ // Apply gain to each channel
30
+ for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
31
+ const channelData = audioBuffer.getChannelData(i);
32
+ const newChannelData = newBuffer.getChannelData(i);
33
+ for (let j = 0; j < channelData.length; j++) {
34
+ // Apply gain and clamp to [-1, 1]
35
+ newChannelData[j] = Math.max(-1, Math.min(1, channelData[j] * gain));
36
+ }
37
+ }
38
+
39
+ return newBuffer;
40
+ }
41
+
42
+ export function convertAudioBufferToWav(audioBuffer) {
43
+ const numOfChannels = audioBuffer.numberOfChannels;
44
+ const length = audioBuffer.length * numOfChannels * 2; // 2 bytes per sample
45
+ const sampleRate = audioBuffer.sampleRate;
46
+ const buffer = new ArrayBuffer(44 + length);
47
+ const view = new DataView(buffer);
48
+
49
+ writeString(view, 0, 'RIFF');
50
+ view.setUint32(4, 36 + length, true);
51
+ writeString(view, 8, 'WAVE');
52
+
53
+ writeString(view, 12, 'fmt ');
54
+ view.setUint32(16, 16, true); // subchunk1 size
55
+ view.setUint16(20, 1, true); // PCM format
56
+ view.setUint16(22, numOfChannels, true);
57
+ view.setUint32(24, sampleRate, true);
58
+ view.setUint32(28, sampleRate * numOfChannels * 2, true); // byte rate
59
+ view.setUint16(32, numOfChannels * 2, true); // block align
60
+ view.setUint16(34, 16, true); // bits per sample
61
+
62
+ writeString(view, 36, 'data');
63
+ view.setUint32(40, length, true);
64
+
65
+ let offset = 44;
66
+ for (let i = 0; i < audioBuffer.numberOfChannels; i++) {
67
+ const channelData = audioBuffer.getChannelData(i);
68
+ for (let j = 0; j < channelData.length; j++) {
69
+ const sample = Math.max(-1, Math.min(1, channelData[j]));
70
+ view.setInt16(offset, sample * 0x7FFF, true);
71
+ offset += 2;
72
+ }
73
+ }
74
+
75
+ return buffer;
76
+ }
77
+
78
+ function writeString(view, offset, string) {
79
+ for (let i = 0; i < string.length; i++) {
80
+ view.setUint8(offset + i, string.charCodeAt(i));
81
+ }
82
+ }
@@ -0,0 +1,198 @@
1
+ import { phonemize as espeakng } from "https://cdn.jsdelivr.net/npm/phonemizer@1.2.1/dist/phonemizer.min.js";
2
+ //import { phonemize as espeakng } from "phonemizer";
3
+
4
+ /**
5
+ * Helper function to split a string on a regex, but keep the delimiters.
6
+ * This is required, because the JavaScript `.split()` method does not keep the delimiters,
7
+ * and wrapping in a capturing group causes issues with existing capturing groups (due to nesting).
8
+ * @param {string} text The text to split.
9
+ * @param {RegExp} regex The regex to split on.
10
+ * @returns {{match: boolean; text: string}[]} The split string.
11
+ */
12
+ function split(text, regex) {
13
+ const result = [];
14
+ let prev = 0;
15
+ for (const match of text.matchAll(regex)) {
16
+ const fullMatch = match[0];
17
+ if (prev < match.index) {
18
+ result.push({ match: false, text: text.slice(prev, match.index) });
19
+ }
20
+ if (fullMatch.length > 0) {
21
+ result.push({ match: true, text: fullMatch });
22
+ }
23
+ prev = match.index + fullMatch.length;
24
+ }
25
+ if (prev < text.length) {
26
+ result.push({ match: false, text: text.slice(prev) });
27
+ }
28
+ return result;
29
+ }
30
+
31
+ /**
32
+ * Helper function to split numbers into phonetic equivalents
33
+ * @param {string} match The matched number
34
+ * @returns {string} The phonetic equivalent
35
+ */
36
+ function split_num(match) {
37
+ if (match.includes(".")) {
38
+ return match;
39
+ } else if (match.includes(":")) {
40
+ let [h, m] = match.split(":").map(Number);
41
+ if (m === 0) {
42
+ return `${h} o'clock`;
43
+ } else if (m < 10) {
44
+ return `${h} oh ${m}`;
45
+ }
46
+ return `${h} ${m}`;
47
+ }
48
+ let year = parseInt(match.slice(0, 4), 10);
49
+ if (year < 1100 || year % 1000 < 10) {
50
+ return match;
51
+ }
52
+ let left = match.slice(0, 2);
53
+ let right = parseInt(match.slice(2, 4), 10);
54
+ let suffix = match.endsWith("s") ? "s" : "";
55
+ if (year % 1000 >= 100 && year % 1000 <= 999) {
56
+ if (right === 0) {
57
+ return `${left} hundred${suffix}`;
58
+ } else if (right < 10) {
59
+ return `${left} oh ${right}${suffix}`;
60
+ }
61
+ }
62
+ return `${left} ${right}${suffix}`;
63
+ }
64
+
65
+ /**
66
+ * Helper function to format monetary values
67
+ * @param {string} match The matched currency
68
+ * @returns {string} The formatted currency
69
+ */
70
+ function flip_money(match) {
71
+ const bill = match[0] === "$" ? "dollar" : "pound";
72
+ if (isNaN(Number(match.slice(1)))) {
73
+ return `${match.slice(1)} ${bill}s`;
74
+ } else if (!match.includes(".")) {
75
+ let suffix = match.slice(1) === "1" ? "" : "s";
76
+ return `${match.slice(1)} ${bill}${suffix}`;
77
+ }
78
+ const [b, c] = match.slice(1).split(".");
79
+ const d = parseInt(c.padEnd(2, "0"), 10);
80
+ let coins = match[0] === "$" ? (d === 1 ? "cent" : "cents") : d === 1 ? "penny" : "pence";
81
+ return `${b} ${bill}${b === "1" ? "" : "s"} and ${d} ${coins}`;
82
+ }
83
+
84
+ /**
85
+ * Helper function to process decimal numbers
86
+ * @param {string} match The matched number
87
+ * @returns {string} The formatted number
88
+ */
89
+ function point_num(match) {
90
+ let [a, b] = match.split(".");
91
+ return `${a} point ${b.split("").join(" ")}`;
92
+ }
93
+
94
+ /**
95
+ * Normalize text for phonemization
96
+ * @param {string} text The text to normalize
97
+ * @returns {string} The normalized text
98
+ */
99
+ function normalize_text(text) {
100
+ return (
101
+ text
102
+ // 1. Handle quotes and brackets
103
+ .replace(/[‘’]/g, "'")
104
+ .replace(/«/g, "“")
105
+ .replace(/»/g, "”")
106
+ .replace(/[“”]/g, '"')
107
+ .replace(/\(/g, "«")
108
+ .replace(/\)/g, "»")
109
+
110
+ // 2. Replace uncommon punctuation marks
111
+ .replace(/、/g, ", ")
112
+ .replace(/。/g, ". ")
113
+ .replace(/!/g, "! ")
114
+ .replace(/,/g, ", ")
115
+ .replace(/:/g, ": ")
116
+ .replace(/;/g, "; ")
117
+ .replace(/?/g, "? ")
118
+
119
+ // 3. Whitespace normalization
120
+ .replace(/[^\S \n]/g, " ")
121
+ .replace(/ +/, " ")
122
+ .replace(/(?<=\n) +(?=\n)/g, "")
123
+
124
+ // 4. Abbreviations
125
+ .replace(/\bD[Rr]\.(?= [A-Z])/g, "Doctor")
126
+ .replace(/\b(?:Mr\.|MR\.(?= [A-Z]))/g, "Mister")
127
+ .replace(/\b(?:Ms\.|MS\.(?= [A-Z]))/g, "Miss")
128
+ .replace(/\b(?:Mrs\.|MRS\.(?= [A-Z]))/g, "Mrs")
129
+ .replace(/\betc\.(?! [A-Z])/gi, "etc")
130
+
131
+ // 5. Normalize casual words
132
+ .replace(/\b(y)eah?\b/gi, "$1e'a")
133
+
134
+ // 5. Handle numbers and currencies
135
+ .replace(/\d*\.\d+|\b\d{4}s?\b|(?<!:)\b(?:[1-9]|1[0-2]):[0-5]\d\b(?!:)/g, split_num)
136
+ .replace(/(?<=\d),(?=\d)/g, "")
137
+ .replace(/[$£]\d+(?:\.\d+)?(?: hundred| thousand| (?:[bm]|tr)illion)*\b|[$£]\d+\.\d\d?\b/gi, flip_money)
138
+ .replace(/\d*\.\d+/g, point_num)
139
+ .replace(/(?<=\d)-(?=\d)/g, " to ")
140
+ .replace(/(?<=\d)S/g, " S")
141
+
142
+ // 6. Handle possessives
143
+ .replace(/(?<=[BCDFGHJ-NP-TV-Z])'?s\b/g, "'S")
144
+ .replace(/(?<=X')S\b/g, "s")
145
+
146
+ // 7. Handle hyphenated words/letters
147
+ .replace(/(?:[A-Za-z]\.){2,} [a-z]/g, (m) => m.replace(/\./g, "-"))
148
+ .replace(/(?<=[A-Z])\.(?=[A-Z])/gi, "-")
149
+
150
+ // 8. Strip leading and trailing whitespace
151
+ .trim()
152
+ );
153
+ }
154
+
155
+ /**
156
+ * Escapes regular expression special characters from a string by replacing them with their escaped counterparts.
157
+ *
158
+ * @param {string} string The string to escape.
159
+ * @returns {string} The escaped string.
160
+ */
161
+ function escapeRegExp(string) {
162
+ return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); // $& means the whole matched string
163
+ }
164
+
165
+ const PUNCTUATION = ';:,.!?¡¿—…"«»“”(){}[]';
166
+ const PUNCTUATION_PATTERN = new RegExp(`(\\s*[${escapeRegExp(PUNCTUATION)}]+\\s*)+`, "g");
167
+
168
+ export async function phonemize(text, language = "a", norm = true) {
169
+ // 1. Normalize text
170
+ if (norm) {
171
+ text = normalize_text(text);
172
+ }
173
+
174
+ // 2. Split into chunks, to ensure we preserve punctuation
175
+ const sections = split(text, PUNCTUATION_PATTERN);
176
+
177
+ // 3. Convert each section to phonemes
178
+ const lang = language === "a" ? "en-us" : "en";
179
+ const ps = (await Promise.all(sections.map(async ({ match, text }) => (match ? text : (await espeakng(text, lang)).join(" "))))).join("");
180
+
181
+ // 4. Post-process phonemes
182
+ let processed = ps
183
+ // https://en.wiktionary.org/wiki/kokoro#English
184
+ .replace(/kəkˈoːɹoʊ/g, "kˈoʊkəɹoʊ")
185
+ .replace(/kəkˈɔːɹəʊ/g, "kˈəʊkəɹəʊ")
186
+ .replace(/ʲ/g, "j")
187
+ .replace(/r/g, "ɹ")
188
+ .replace(/x/g, "k")
189
+ .replace(/ɬ/g, "l")
190
+ .replace(/(?<=[a-zɹː])(?=hˈʌndɹɪd)/g, " ")
191
+ .replace(/ z(?=[;:,.!?¡¿—…"«»“” ]|$)/g, "z");
192
+
193
+ // 5. Additional post-processing for American English
194
+ if (language === "a") {
195
+ processed = processed.replace(/(?<=nˈaɪn)ti(?!ː)/g, "di");
196
+ }
197
+ return processed.trim();
198
+ }
@@ -0,0 +1,107 @@
1
+ export function splitTextSmart(text, maxChunkLength = 500) {
2
+ const paragraphChunks = text.split(/\n\s*\n/);
3
+ const finalChunks = [];
4
+
5
+ for (let para of paragraphChunks) {
6
+ if (para.length <= maxChunkLength) {
7
+ finalChunks.push(para.trim());
8
+ continue;
9
+ }
10
+
11
+ const sentenceRegex = /(?<=[.?!])(?=\s+["“”'a-z])/gi;
12
+ const sentences = para.split(sentenceRegex);
13
+
14
+ let chunk = '';
15
+ for (let sentence of sentences) {
16
+ sentence = sentence.trim();
17
+
18
+ if (sentence.length > maxChunkLength) {
19
+ // Sentence too long — fallback split
20
+ const subChunks = splitLongSentence(sentence, maxChunkLength);
21
+ for (let sub of subChunks) {
22
+ if ((chunk + ' ' + sub).length > maxChunkLength) {
23
+ if (chunk) finalChunks.push(chunk.trim());
24
+ chunk = sub;
25
+ } else {
26
+ chunk += (chunk ? ' ' : '') + sub;
27
+ }
28
+ }
29
+ continue;
30
+ }
31
+
32
+ if ((chunk + ' ' + sentence).length > maxChunkLength) {
33
+ if (chunk) finalChunks.push(chunk.trim());
34
+ chunk = sentence;
35
+ } else {
36
+ chunk += (chunk ? ' ' : '') + sentence;
37
+ }
38
+ }
39
+ if (chunk) finalChunks.push(chunk.trim());
40
+ }
41
+
42
+ return finalChunks;
43
+ }
44
+
45
+ export function splitLongSentence(sentence, maxLen) {
46
+ const chunks = [];
47
+ let current = '';
48
+
49
+ const commaParts = sentence.split(/,\s*/);
50
+ for (let part of commaParts) {
51
+ if ((current + ', ' + part).length > maxLen) {
52
+ if (current) chunks.push(current.trim());
53
+ if (part.length > maxLen) {
54
+ const words = part.split(/\s+/);
55
+ let wordChunk = '';
56
+ for (let word of words) {
57
+ if ((wordChunk + ' ' + word).length > maxLen) {
58
+ if (wordChunk) chunks.push(wordChunk.trim());
59
+ wordChunk = word;
60
+ } else {
61
+ wordChunk += (wordChunk ? ' ' : '') + word;
62
+ }
63
+ }
64
+ if (wordChunk) chunks.push(wordChunk.trim());
65
+ current = '';
66
+ } else {
67
+ current = part;
68
+ }
69
+ } else {
70
+ current += (current ? ', ' : '') + part;
71
+ }
72
+ }
73
+ if (current) chunks.push(current.trim());
74
+
75
+ return chunks;
76
+ }
77
+
78
+
79
+ function splitTextSmartOld(text, maxChunkLength = 500) {
80
+ const paragraphChunks = text.split(/\n\s*\n/); // Step 1: split on double returns
81
+ const finalChunks = [];
82
+
83
+ for (let para of paragraphChunks) {
84
+ if (para.length <= maxChunkLength) {
85
+ finalChunks.push(para.trim());
86
+ continue;
87
+ }
88
+
89
+ // Step 2: Further split on sentence boundaries if too long
90
+ const sentenceRegex = /(?<=[.?!])(?=\s+["“”'a-z])/gi;
91
+ const sentences = para.split(sentenceRegex);
92
+
93
+ let chunk = '';
94
+ for (let sentence of sentences) {
95
+ sentence = sentence.trim();
96
+ if ((chunk + ' ' + sentence).length > maxChunkLength) {
97
+ if (chunk) finalChunks.push(chunk.trim());
98
+ chunk = sentence;
99
+ } else {
100
+ chunk += (chunk ? ' ' : '') + sentence;
101
+ }
102
+ }
103
+ if (chunk) finalChunks.push(chunk.trim());
104
+ }
105
+
106
+ return finalChunks;
107
+ }
@@ -0,0 +1,89 @@
1
+
2
+ /**
3
+ * Utility functions for detecting sentence boundaries in streaming text
4
+ */
5
+
6
+ // These are common sentence ending patterns
7
+ const SENTENCE_PATTERNS = {
8
+ // Basic sentence endings: period, exclamation, question mark followed by space or end
9
+ basicEnd: /[.!?][\s"')\]]*($|(?=\s*[A-Z]))/,
10
+
11
+ // Dialog endings: quote followed by punctuation and space or end
12
+ dialogEnd: /["'][,.!?][\s"')\]]*($|(?=\s*[A-Z]))/,
13
+
14
+ // List item or enumeration endings: semicolon, colon
15
+ listItemEnd: /[;:][\s]*($|(?=\s*[-•*]))/,
16
+
17
+ // Paragraph breaks: double newline
18
+ paragraphBreak: /\n\s*\n/,
19
+
20
+ // Force breaks for very long text without clear sentence boundaries
21
+ longText: (text) => text.length > 150
22
+ };
23
+
24
+ /**
25
+ * Determines if a text contains a complete sentence or should be sent for TTS
26
+ * @param {string} text - The text to check
27
+ * @param {Object} options - Configuration options
28
+ * @returns {boolean} - True if the text contains a complete sentence or should be spoken
29
+ */
30
+ export function isCompleteSentence(text) {
31
+ if (!text || text.trim().length === 0) {
32
+ return false;
33
+ }
34
+
35
+ // Check all patterns
36
+ return (
37
+ SENTENCE_PATTERNS.basicEnd.test(text) ||
38
+ SENTENCE_PATTERNS.dialogEnd.test(text) ||
39
+ SENTENCE_PATTERNS.listItemEnd.test(text) ||
40
+ SENTENCE_PATTERNS.paragraphBreak.test(text) ||
41
+ SENTENCE_PATTERNS.longText(text)
42
+ );
43
+ }
44
+
45
+ /**
46
+ * Process streaming text and return complete sentences
47
+ *
48
+ * @param {string} accumulator - The text accumulated so far
49
+ * @param {string} newContent - The new content to add
50
+ * @returns {Object} - Object with processed results:
51
+ * - sentences: Array of complete sentences to speak
52
+ * - remainder: Remaining text that doesn't form a complete sentence yet
53
+ */
54
+ export function processStreamingText(accumulator, newContent) {
55
+ const combinedText = accumulator + newContent;
56
+
57
+ // If the text is empty, just return
58
+ if (!combinedText || combinedText.trim().length === 0) {
59
+ return { sentences: [], remainder: '' };
60
+ }
61
+
62
+ // Split text by common sentence boundaries to check for multiple sentences
63
+ const parts = combinedText.split(/(?<=[.!?][\s"')\]])/);
64
+
65
+ // If we only have one part and it's not a complete sentence
66
+ if (parts.length === 1 && !isCompleteSentence(parts[0])) {
67
+ return { sentences: [], remainder: combinedText };
68
+ }
69
+
70
+ // Otherwise process all parts
71
+ const sentences = [];
72
+ let currentSentence = '';
73
+
74
+ for (let i = 0; i < parts.length; i++) {
75
+ currentSentence += parts[i];
76
+
77
+ // If this forms a complete sentence or is the last part that should be spoken
78
+ if (isCompleteSentence(currentSentence) ||
79
+ (i === parts.length - 1 && SENTENCE_PATTERNS.longText(currentSentence))) {
80
+ sentences.push(currentSentence.trim());
81
+ currentSentence = '';
82
+ }
83
+ }
84
+
85
+ return {
86
+ sentences,
87
+ remainder: currentSentence
88
+ };
89
+ }
package/readme.md DELETED
@@ -1 +0,0 @@
1
- ![logo](https://i.imgur.com/ypVzqbg.png)