claude-phone-local 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +108 -0
- package/Dockerfile +84 -0
- package/LICENSE +22 -0
- package/README.md +231 -0
- package/claude-api-server/package-lock.json +832 -0
- package/claude-api-server/package.json +20 -0
- package/claude-api-server/server.js +540 -0
- package/claude-api-server/structured.js +208 -0
- package/cli/README.md +231 -0
- package/cli/bin/check-publish-safety.js +53 -0
- package/cli/bin/claude-phone.js +23 -0
- package/cli/bin/cli-main.js +284 -0
- package/cli/bin/postinstall.js +20 -0
- package/cli/lib/commands/api-server.js +77 -0
- package/cli/lib/commands/backup.js +69 -0
- package/cli/lib/commands/config/path.js +30 -0
- package/cli/lib/commands/config/reset.js +73 -0
- package/cli/lib/commands/config/show.js +86 -0
- package/cli/lib/commands/device/add.js +143 -0
- package/cli/lib/commands/device/list.js +55 -0
- package/cli/lib/commands/device/remove.js +83 -0
- package/cli/lib/commands/doctor.js +422 -0
- package/cli/lib/commands/logs.js +186 -0
- package/cli/lib/commands/restore.js +172 -0
- package/cli/lib/commands/setup.js +1246 -0
- package/cli/lib/commands/start.js +346 -0
- package/cli/lib/commands/status.js +139 -0
- package/cli/lib/commands/stop.js +101 -0
- package/cli/lib/commands/uninstall.js +183 -0
- package/cli/lib/commands/update.js +205 -0
- package/cli/lib/config.js +113 -0
- package/cli/lib/docker.js +384 -0
- package/cli/lib/mcp-register.js +108 -0
- package/cli/lib/network.js +109 -0
- package/cli/lib/platform.js +85 -0
- package/cli/lib/port-check.js +135 -0
- package/cli/lib/prereqs/checks/compose.js +147 -0
- package/cli/lib/prereqs/checks/disk.js +89 -0
- package/cli/lib/prereqs/checks/docker.js +155 -0
- package/cli/lib/prereqs/checks/network.js +78 -0
- package/cli/lib/prereqs/checks/node.js +95 -0
- package/cli/lib/prereqs/installers/docker-desktop.js +248 -0
- package/cli/lib/prereqs/installers/docker.js +229 -0
- package/cli/lib/prereqs/installers/node.js +254 -0
- package/cli/lib/prereqs/platform.js +175 -0
- package/cli/lib/prereqs/utils/execute.js +208 -0
- package/cli/lib/prereqs/utils/rollback.js +223 -0
- package/cli/lib/prereqs/utils/sudo.js +177 -0
- package/cli/lib/prereqs.js +228 -0
- package/cli/lib/prerequisites.js +115 -0
- package/cli/lib/process-manager.js +176 -0
- package/cli/lib/utils.js +78 -0
- package/cli/lib/validators.js +219 -0
- package/cli/lib/voice-downloader.js +79 -0
- package/docker/entrypoint.sh +110 -0
- package/docker/supervisord.conf +78 -0
- package/docker-compose.yml +40 -0
- package/docs/CLAUDE-CODE-SKILL.md +45 -0
- package/docs/LANGUAGES.md +113 -0
- package/docs/MCP-SERVER.md +111 -0
- package/docs/TROUBLESHOOTING.md +288 -0
- package/mcp-server/index.js +159 -0
- package/mcp-server/package-lock.json +1201 -0
- package/mcp-server/package.json +9 -0
- package/package.json +73 -0
- package/stt-local/Dockerfile +12 -0
- package/stt-local/server.py +45 -0
- package/tts-local/Dockerfile +9 -0
- package/tts-local/server.py +83 -0
- package/voice-app/API-QUERY-CONTRACT.md +238 -0
- package/voice-app/DEPLOYMENT.md +176 -0
- package/voice-app/Dockerfile +20 -0
- package/voice-app/README-OUTBOUND.md +316 -0
- package/voice-app/config/devices.json.example +18 -0
- package/voice-app/index.js +273 -0
- package/voice-app/lib/audio-fork.js +465 -0
- package/voice-app/lib/claude-bridge.js +113 -0
- package/voice-app/lib/connection-retry.js +58 -0
- package/voice-app/lib/conversation-loop.js +490 -0
- package/voice-app/lib/device-registry.js +171 -0
- package/voice-app/lib/http-server.js +242 -0
- package/voice-app/lib/logger.js +35 -0
- package/voice-app/lib/multi-registrar.js +133 -0
- package/voice-app/lib/outbound-handler.js +299 -0
- package/voice-app/lib/outbound-routes.js +417 -0
- package/voice-app/lib/outbound-session.js +324 -0
- package/voice-app/lib/query-routes.js +484 -0
- package/voice-app/lib/registrar.js +144 -0
- package/voice-app/lib/sip-handler.js +490 -0
- package/voice-app/lib/tts-service.js +205 -0
- package/voice-app/lib/whisper-client.js +140 -0
- package/voice-app/package.json +38 -0
- package/voice-app/static/gotit-beep.wav +0 -0
- package/voice-app/static/hold-music.wav +0 -0
- package/voice-app/static/ready-beep.wav +0 -0
- package/voice-app/test/freeswitch-retry.test.js +100 -0
|
@@ -0,0 +1,490 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SIP Call Handler with Conversation Loop
|
|
3
|
+
* v12: Device registry integration with proper method names
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
const { setTimeout: sleep } = require('node:timers/promises');
|
|
7
|
+
|
|
8
|
+
// FreeSWITCH (a separate container) fetches/connects to these - must be the
|
|
9
|
+
// voice-app container's address, not FreeSWITCH's own loopback.
|
|
10
|
+
const AUDIO_BASE_URL = process.env.AUDIO_BASE_URL || 'http://voice-app:3000';
|
|
11
|
+
const AUDIO_WS_HOST = new URL(AUDIO_BASE_URL).hostname;
|
|
12
|
+
|
|
13
|
+
// Audio cue URLs
|
|
14
|
+
const READY_BEEP_URL = `${AUDIO_BASE_URL}/static/ready-beep.wav`;
|
|
15
|
+
const GOTIT_BEEP_URL = `${AUDIO_BASE_URL}/static/gotit-beep.wav`;
|
|
16
|
+
const HOLD_MUSIC_URL = `${AUDIO_BASE_URL}/static/hold-music.wav`;
|
|
17
|
+
|
|
18
|
+
// Default voice ID (Morpheus)
|
|
19
|
+
const DEFAULT_VOICE_ID = 'JAgnJveGGUh4qy4kh6dF';
|
|
20
|
+
|
|
21
|
+
// Whisper language code -> Piper voice installed in tts-local/voices/.
|
|
22
|
+
// Whisper detects the caller's language per utterance and we answer in the
|
|
23
|
+
// same one. Override/extend with LANG_VOICE_MAP in .env as JSON.
|
|
24
|
+
const DEFAULT_LANG_VOICES = {
|
|
25
|
+
en: 'en_US-lessac-medium',
|
|
26
|
+
hi: 'hi_IN-priyamvada-medium',
|
|
27
|
+
mr: 'mr_IN-google-medium'
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
let LANG_VOICES = DEFAULT_LANG_VOICES;
|
|
31
|
+
if (process.env.LANG_VOICE_MAP) {
|
|
32
|
+
try {
|
|
33
|
+
LANG_VOICES = Object.assign({}, DEFAULT_LANG_VOICES, JSON.parse(process.env.LANG_VOICE_MAP));
|
|
34
|
+
} catch (e) {
|
|
35
|
+
console.log('[' + new Date().toISOString() + '] LANG: bad LANG_VOICE_MAP JSON, using defaults');
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Languages we will actually answer in. Anything Whisper detects outside this
|
|
40
|
+
// set falls back to the device's own voice, so a misdetection can't leave us
|
|
41
|
+
// with no installed model.
|
|
42
|
+
const SUPPORTED_LANGS = (process.env.SUPPORTED_LANGS || 'en,hi,mr')
|
|
43
|
+
.split(',').map(function (x) { return x.trim(); }).filter(Boolean);
|
|
44
|
+
|
|
45
|
+
function voiceForLanguage(lang, fallbackVoice) {
|
|
46
|
+
if (!lang) return fallbackVoice;
|
|
47
|
+
if (SUPPORTED_LANGS.indexOf(lang) === -1) return fallbackVoice;
|
|
48
|
+
return LANG_VOICES[lang] || fallbackVoice;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// Claude Code-style thinking phrases
|
|
52
|
+
const THINKING_PHRASES = [
|
|
53
|
+
"Let me check that for you.",
|
|
54
|
+
"One moment.",
|
|
55
|
+
"Just a second.",
|
|
56
|
+
"Looking into it now.",
|
|
57
|
+
"Give me a moment.",
|
|
58
|
+
"Checking on that.",
|
|
59
|
+
"Hang on, almost there.",
|
|
60
|
+
"Still working on it.",
|
|
61
|
+
"Nearly done.",
|
|
62
|
+
"Bear with me a moment.",
|
|
63
|
+
];
|
|
64
|
+
|
|
65
|
+
// Said only while the caller is already waiting, so they never hear the same
|
|
66
|
+
// line twice in a row within one wait.
|
|
67
|
+
const WAITING_PHRASES = [
|
|
68
|
+
"Still working on this.",
|
|
69
|
+
"Almost there.",
|
|
70
|
+
"Just a little longer.",
|
|
71
|
+
"Nearly finished.",
|
|
72
|
+
"Hang in there, still going.",
|
|
73
|
+
"Won't be much longer.",
|
|
74
|
+
];
|
|
75
|
+
|
|
76
|
+
function getRandomThinkingPhrase() {
|
|
77
|
+
return THINKING_PHRASES[Math.floor(Math.random() * THINKING_PHRASES.length)];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function extractCallerId(req) {
|
|
81
|
+
var from = req.get("From") || "";
|
|
82
|
+
var match = from.match(/sip:([+\d]+)@/);
|
|
83
|
+
if (match) return match[1];
|
|
84
|
+
var numMatch = from.match(/<sip:(\d+)@/);
|
|
85
|
+
if (numMatch) return numMatch[1];
|
|
86
|
+
return "unknown";
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Extract dialed extension from SIP To header
|
|
91
|
+
*/
|
|
92
|
+
function extractDialedExtension(req) {
|
|
93
|
+
var to = req.get("To") || "";
|
|
94
|
+
var match = to.match(/sip:(\d+)@/);
|
|
95
|
+
if (match) {
|
|
96
|
+
return match[1];
|
|
97
|
+
}
|
|
98
|
+
return null;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function isGoodbye(transcript) {
|
|
102
|
+
const lower = transcript.toLowerCase().trim();
|
|
103
|
+
const goodbyePhrases = [
|
|
104
|
+
// English
|
|
105
|
+
'goodbye', 'good bye', 'bye', 'bye bye', 'hang up', 'end call', 'end the call',
|
|
106
|
+
'close the call', 'cut the call', 'disconnect', "that's all", 'thats all',
|
|
107
|
+
'thank you bye', 'talk later',
|
|
108
|
+
// Hindi / Marathi (Devanagari + common romanisations)
|
|
109
|
+
'अलविदा', 'नमस्ते', 'बाय', 'फोन बंद करो', 'कॉल बंद करो', 'बंद करो',
|
|
110
|
+
'ठेवतो', 'ठेवते', 'फोन ठेव', 'बंद कर',
|
|
111
|
+
'alvida', 'phone band karo', 'call band karo', 'band karo', 'thevto'
|
|
112
|
+
];
|
|
113
|
+
return goodbyePhrases.some(function(phrase) {
|
|
114
|
+
return lower === phrase || lower.includes(' ' + phrase) ||
|
|
115
|
+
lower.startsWith(phrase + ' ') || lower.endsWith(' ' + phrase);
|
|
116
|
+
});
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Extract voice-friendly line from Claude's response
|
|
121
|
+
* Priority: VOICE_RESPONSE > CUSTOM COMPLETED > COMPLETED > first sentence
|
|
122
|
+
*/
|
|
123
|
+
function extractVoiceLine(response) {
|
|
124
|
+
// Priority 1: VOICE_RESPONSE (new format)
|
|
125
|
+
var voiceMatch = response.match(/🗣️\s*VOICE_RESPONSE:\s*([^\n]+)/im);
|
|
126
|
+
if (voiceMatch) {
|
|
127
|
+
var text = voiceMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
|
|
128
|
+
if (text && text.split(/\s+/).length <= 60) {
|
|
129
|
+
return text;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// Priority 2: CUSTOM COMPLETED
|
|
134
|
+
var customMatch = response.match(/🗣️\s*CUSTOM\s+COMPLETED:\s*(.+?)(?:\n|$)/im);
|
|
135
|
+
if (customMatch) {
|
|
136
|
+
text = customMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
|
|
137
|
+
if (text && text.split(/\s+/).length <= 50) {
|
|
138
|
+
return text;
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// Priority 3: COMPLETED
|
|
143
|
+
var completedMatch = response.match(/🎯\s*COMPLETED:\s*(.+?)(?:\n|$)/im);
|
|
144
|
+
if (completedMatch) {
|
|
145
|
+
return completedMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// Priority 4: First sentence
|
|
149
|
+
var firstSentence = response.split(/[.!?]/)[0];
|
|
150
|
+
if (firstSentence && firstSentence.length < 500) {
|
|
151
|
+
return firstSentence.trim();
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
return response.substring(0, 500).trim();
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Play a clip the caller is allowed to interrupt.
|
|
159
|
+
*
|
|
160
|
+
* FreeSWITCH plays to completion unless told otherwise, so to support barge-in
|
|
161
|
+
* we arm the detector, then issue uuid_break the moment the caller starts
|
|
162
|
+
* talking. Returns true if the caller cut in, so the loop can skip straight to
|
|
163
|
+
* listening instead of finishing what it was saying.
|
|
164
|
+
*/
|
|
165
|
+
async function playInterruptible(endpoint, session, url) {
|
|
166
|
+
if (!session) {
|
|
167
|
+
await endpoint.play(url);
|
|
168
|
+
return false;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
let barged = false;
|
|
172
|
+
const onBarge = function () {
|
|
173
|
+
barged = true;
|
|
174
|
+
// uuid_break stops the current playback on this leg immediately.
|
|
175
|
+
endpoint.api('uuid_break', endpoint.uuid).catch(function () {});
|
|
176
|
+
};
|
|
177
|
+
|
|
178
|
+
session.once('barge-in', onBarge);
|
|
179
|
+
session.setBargeInEnabled(true);
|
|
180
|
+
try {
|
|
181
|
+
await endpoint.play(url);
|
|
182
|
+
} finally {
|
|
183
|
+
session.setBargeInEnabled(false);
|
|
184
|
+
session.removeListener('barge-in', onBarge);
|
|
185
|
+
}
|
|
186
|
+
if (barged) {
|
|
187
|
+
console.log('[' + new Date().toISOString() + '] BARGE-IN: caller interrupted, listening now');
|
|
188
|
+
}
|
|
189
|
+
return barged;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Main conversation loop
|
|
194
|
+
* @param {Object} deviceConfig - Device configuration (name, prompt, voiceId, etc.) or null for default
|
|
195
|
+
*/
|
|
196
|
+
async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfig) {
|
|
197
|
+
const { ttsService, whisperClient, claudeBridge, wsPort, audioForkServer } = options;
|
|
198
|
+
|
|
199
|
+
let session = null;
|
|
200
|
+
let forkRunning = false;
|
|
201
|
+
|
|
202
|
+
// Get device-specific settings
|
|
203
|
+
const deviceName = deviceConfig ? deviceConfig.name : 'Morpheus';
|
|
204
|
+
const devicePrompt = deviceConfig ? deviceConfig.prompt : null;
|
|
205
|
+
// Local mode: device.voice is a Piper voice name (null falls back to PIPER_VOICE env).
|
|
206
|
+
// Cloud mode: device.voiceId is an ElevenLabs voice ID (falls back to DEFAULT_VOICE_ID).
|
|
207
|
+
const voiceId = ttsService.mode === 'cloud'
|
|
208
|
+
? ((deviceConfig && deviceConfig.voiceId) ? deviceConfig.voiceId : DEFAULT_VOICE_ID)
|
|
209
|
+
: ((deviceConfig && deviceConfig.voice) ? deviceConfig.voice : null);
|
|
210
|
+
// Voice used for the current turn - starts as the device voice and follows
|
|
211
|
+
// the caller's detected language from the first utterance onward.
|
|
212
|
+
let turnVoice = ttsService.mode === 'cloud'
|
|
213
|
+
? (deviceConfig && deviceConfig.voiceId ? deviceConfig.voiceId : DEFAULT_VOICE_ID)
|
|
214
|
+
: (deviceConfig && deviceConfig.voice ? deviceConfig.voice : null);
|
|
215
|
+
|
|
216
|
+
const greeting = deviceConfig && deviceConfig.name !== 'Morpheus'
|
|
217
|
+
? "Hello! I'm " + deviceConfig.name + ". How can I help you today?"
|
|
218
|
+
: "Hello! I'm your server. How can I help you today?";
|
|
219
|
+
|
|
220
|
+
try {
|
|
221
|
+
console.log('[' + new Date().toISOString() + '] CONVERSATION Starting (session: ' + callUuid + ', device: ' + deviceName + ', voice: ' + voiceId + ')...');
|
|
222
|
+
|
|
223
|
+
// Play device-specific greeting with device voice BEFORE starting audio fork
|
|
224
|
+
console.log('[' + new Date().toISOString() + '] Generating greeting...');
|
|
225
|
+
const greetingUrl = await ttsService.generateSpeech(greeting, voiceId);
|
|
226
|
+
console.log('[' + new Date().toISOString() + '] Playing greeting: ' + greetingUrl);
|
|
227
|
+
await endpoint.play(greetingUrl);
|
|
228
|
+
console.log('[' + new Date().toISOString() + '] Greeting played successfully');
|
|
229
|
+
|
|
230
|
+
// Start fork for entire call AFTER greeting
|
|
231
|
+
const wsUrl = 'ws://' + AUDIO_WS_HOST + ':' + wsPort + '/' + encodeURIComponent(callUuid);
|
|
232
|
+
const sessionPromise = audioForkServer.expectSession(callUuid, { timeoutMs: 10000 });
|
|
233
|
+
|
|
234
|
+
await endpoint.forkAudioStart({
|
|
235
|
+
wsUrl: wsUrl,
|
|
236
|
+
mixType: 'mono',
|
|
237
|
+
sampling: '16k'
|
|
238
|
+
});
|
|
239
|
+
forkRunning = true;
|
|
240
|
+
|
|
241
|
+
session = await sessionPromise;
|
|
242
|
+
console.log('[' + new Date().toISOString() + '] AUDIO Fork connected');
|
|
243
|
+
|
|
244
|
+
// Main conversation loop
|
|
245
|
+
let turnCount = 0;
|
|
246
|
+
const MAX_TURNS = 20;
|
|
247
|
+
|
|
248
|
+
while (turnCount < MAX_TURNS) {
|
|
249
|
+
turnCount++;
|
|
250
|
+
console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + '/' + MAX_TURNS);
|
|
251
|
+
|
|
252
|
+
// READY BEEP
|
|
253
|
+
try {
|
|
254
|
+
await endpoint.play(READY_BEEP_URL);
|
|
255
|
+
} catch (e) {
|
|
256
|
+
console.log('[' + new Date().toISOString() + '] BEEP: Ready beep failed, continuing');
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
session.setCaptureEnabled(true);
|
|
260
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Waiting for speech...');
|
|
261
|
+
|
|
262
|
+
let utterance = null;
|
|
263
|
+
try {
|
|
264
|
+
utterance = await session.waitForUtterance({ timeoutMs: 30000 });
|
|
265
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Got: ' + utterance.audio.length + ' bytes');
|
|
266
|
+
} catch (err) {
|
|
267
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Timeout: ' + err.message);
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
session.setCaptureEnabled(false);
|
|
271
|
+
|
|
272
|
+
if (!utterance) {
|
|
273
|
+
const promptUrl = await ttsService.generateSpeech("I didn't hear anything. Are you still there?", turnVoice);
|
|
274
|
+
await endpoint.play(promptUrl);
|
|
275
|
+
continue;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// GOT-IT BEEP
|
|
279
|
+
try {
|
|
280
|
+
await endpoint.play(GOTIT_BEEP_URL);
|
|
281
|
+
} catch (e) {
|
|
282
|
+
console.log('[' + new Date().toISOString() + '] BEEP: Got-it beep failed, continuing');
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
// Transcribe (language auto-detected unless STT_LANGUAGE pins one)
|
|
286
|
+
const sttResult = await whisperClient.transcribeDetailed(utterance.audio, {
|
|
287
|
+
format: 'pcm',
|
|
288
|
+
sampleRate: 16000,
|
|
289
|
+
language: (deviceConfig && deviceConfig.language) || process.env.STT_LANGUAGE || 'auto'
|
|
290
|
+
});
|
|
291
|
+
const transcript = sttResult.text;
|
|
292
|
+
const detectedLang = sttResult.language;
|
|
293
|
+
|
|
294
|
+
// Answer in whatever language the caller just used.
|
|
295
|
+
turnVoice = voiceForLanguage(detectedLang, voiceId);
|
|
296
|
+
console.log('[' + new Date().toISOString() + '] WHISPER [' + (detectedLang || '?') +
|
|
297
|
+
' -> voice ' + turnVoice + ']: "' + transcript + '"');
|
|
298
|
+
|
|
299
|
+
if (!transcript || transcript.trim().length < 2) {
|
|
300
|
+
const clarifyUrl = await ttsService.generateSpeech("Sorry, I didn't catch that. Could you repeat?", turnVoice);
|
|
301
|
+
await endpoint.play(clarifyUrl);
|
|
302
|
+
continue;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
if (isGoodbye(transcript)) {
|
|
306
|
+
const byeUrl = await ttsService.generateSpeech("Goodbye! Call again anytime.", turnVoice);
|
|
307
|
+
await endpoint.play(byeUrl);
|
|
308
|
+
break;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
// THINKING FEEDBACK
|
|
312
|
+
const thinkingPhrase = getRandomThinkingPhrase();
|
|
313
|
+
console.log('[' + new Date().toISOString() + '] THINKING: "' + thinkingPhrase + '"');
|
|
314
|
+
const thinkingUrl = await ttsService.generateSpeech(thinkingPhrase, turnVoice);
|
|
315
|
+
await endpoint.play(thinkingUrl);
|
|
316
|
+
|
|
317
|
+
// hears nothing for the whole query and hangs up.
|
|
318
|
+
// Fill the whole wait, not just parts of it. The gap alternates between a
|
|
319
|
+
// soft music bed and a spoken line, so the line never goes dead. Clips play
|
|
320
|
+
// to completion - one endpoint cannot layer two streams - and every Nth
|
|
321
|
+
// round is speech instead of music.
|
|
322
|
+
let waiting = true;
|
|
323
|
+
const SPEAK_EVERY = parseInt(process.env.KEEPALIVE_SPEAK_EVERY || '3', 10);
|
|
324
|
+
const keepAlive = (async function () {
|
|
325
|
+
let round = 0;
|
|
326
|
+
while (waiting) {
|
|
327
|
+
round++;
|
|
328
|
+
try {
|
|
329
|
+
const clipUrl = (round % SPEAK_EVERY === 0)
|
|
330
|
+
? await ttsService.generateSpeech(getRandomWaitingPhrase(), turnVoice)
|
|
331
|
+
: HOLD_MUSIC_URL;
|
|
332
|
+
if (!waiting) break;
|
|
333
|
+
// Caller can cut through the hold music / filler to add something.
|
|
334
|
+
if (await playInterruptible(endpoint, session, clipUrl)) {
|
|
335
|
+
waiting = false;
|
|
336
|
+
break;
|
|
337
|
+
}
|
|
338
|
+
} catch (e) {
|
|
339
|
+
console.log('[' + new Date().toISOString() + '] KEEPALIVE: stopped (' + e.message + ')');
|
|
340
|
+
return;
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
})();
|
|
344
|
+
|
|
345
|
+
// Query Claude with device-specific prompt
|
|
346
|
+
console.log('[' + new Date().toISOString() + '] CLAUDE Querying (device: ' + deviceName + ')...');
|
|
347
|
+
let claudeResponse;
|
|
348
|
+
try {
|
|
349
|
+
claudeResponse = await claudeBridge.query(
|
|
350
|
+
transcript,
|
|
351
|
+
{ callId: callUuid, devicePrompt: devicePrompt }
|
|
352
|
+
);
|
|
353
|
+
} finally {
|
|
354
|
+
waiting = false;
|
|
355
|
+
try { await keepAlive; } catch (e) {}
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
console.log('[' + new Date().toISOString() + '] CLAUDE Response received');
|
|
359
|
+
|
|
360
|
+
// Extract and play voice line with device voice
|
|
361
|
+
const voiceLine = extractVoiceLine(claudeResponse);
|
|
362
|
+
console.log('[' + new Date().toISOString() + '] VOICE: "' + voiceLine + '"');
|
|
363
|
+
|
|
364
|
+
const responseUrl = await ttsService.generateSpeech(voiceLine, turnVoice);
|
|
365
|
+
// Long answers are the usual thing people want to interrupt.
|
|
366
|
+
await playInterruptible(endpoint, session, responseUrl);
|
|
367
|
+
|
|
368
|
+
console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + ' complete');
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
if (turnCount >= MAX_TURNS) {
|
|
372
|
+
const maxUrl = await ttsService.generateSpeech("We've been talking for a while. Goodbye!", turnVoice);
|
|
373
|
+
await endpoint.play(maxUrl);
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
} catch (error) {
|
|
377
|
+
console.error('[' + new Date().toISOString() + '] CONVERSATION Error:', error.message);
|
|
378
|
+
try {
|
|
379
|
+
if (session) session.setCaptureEnabled(false);
|
|
380
|
+
const errUrl = await ttsService.generateSpeech("Sorry, something went wrong.", voiceId);
|
|
381
|
+
await endpoint.play(errUrl);
|
|
382
|
+
} catch (e) {}
|
|
383
|
+
} finally {
|
|
384
|
+
console.log('[' + new Date().toISOString() + '] CONVERSATION Cleanup...');
|
|
385
|
+
|
|
386
|
+
try {
|
|
387
|
+
await claudeBridge.endSession(callUuid);
|
|
388
|
+
} catch (e) {}
|
|
389
|
+
|
|
390
|
+
if (forkRunning) {
|
|
391
|
+
try {
|
|
392
|
+
await endpoint.forkAudioStop();
|
|
393
|
+
} catch (e) {}
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
try { dialog.destroy(); } catch (e) {}
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* Strip video tracks from SDP (FreeSWITCH doesn't support H.261 and rejects with 488)
|
|
402
|
+
* Keeps only audio tracks to ensure codec negotiation succeeds
|
|
403
|
+
*/
|
|
404
|
+
function stripVideoFromSdp(sdp) {
|
|
405
|
+
if (!sdp) return sdp;
|
|
406
|
+
|
|
407
|
+
const lines = sdp.split('\r\n');
|
|
408
|
+
const result = [];
|
|
409
|
+
let inVideoSection = false;
|
|
410
|
+
|
|
411
|
+
for (const line of lines) {
|
|
412
|
+
// Check if we're entering a video media section
|
|
413
|
+
if (line.startsWith('m=video')) {
|
|
414
|
+
inVideoSection = true;
|
|
415
|
+
continue; // Skip the m=video line
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
// Check if we're entering a new media section (audio, etc.)
|
|
419
|
+
if (line.startsWith('m=') && !line.startsWith('m=video')) {
|
|
420
|
+
inVideoSection = false;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
// Skip all lines in the video section
|
|
424
|
+
if (inVideoSection) {
|
|
425
|
+
continue;
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
result.push(line);
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
return result.join('\r\n');
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
/**
|
|
435
|
+
* Handle incoming SIP INVITE
|
|
436
|
+
*/
|
|
437
|
+
async function handleInvite(req, res, options) {
|
|
438
|
+
const { mediaServer, deviceRegistry } = options;
|
|
439
|
+
|
|
440
|
+
const callerId = extractCallerId(req);
|
|
441
|
+
const dialedExt = extractDialedExtension(req);
|
|
442
|
+
|
|
443
|
+
// Look up device config using deviceRegistry.get() (works with name OR extension)
|
|
444
|
+
let deviceConfig = null;
|
|
445
|
+
if (deviceRegistry && dialedExt) {
|
|
446
|
+
deviceConfig = deviceRegistry.get(dialedExt);
|
|
447
|
+
if (deviceConfig) {
|
|
448
|
+
console.log('[' + new Date().toISOString() + '] CALL Device matched: ' + deviceConfig.name + ' (ext ' + dialedExt + ')');
|
|
449
|
+
} else {
|
|
450
|
+
console.log('[' + new Date().toISOString() + '] CALL Unknown extension ' + dialedExt + ', using default');
|
|
451
|
+
deviceConfig = deviceRegistry.getDefault();
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
console.log('[' + new Date().toISOString() + '] CALL Incoming from: ' + callerId + ' to ext: ' + (dialedExt || 'unknown'));
|
|
456
|
+
|
|
457
|
+
try {
|
|
458
|
+
// Strip video from SDP to avoid FreeSWITCH 488 error with unsupported video codecs
|
|
459
|
+
const originalSdp = req.body;
|
|
460
|
+
const audioOnlySdp = stripVideoFromSdp(originalSdp);
|
|
461
|
+
if (originalSdp !== audioOnlySdp) {
|
|
462
|
+
console.log('[' + new Date().toISOString() + '] CALL Stripped video track from SDP');
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
const result = await mediaServer.connectCaller(req, res, { remoteSdp: audioOnlySdp });
|
|
466
|
+
const { endpoint, dialog } = result;
|
|
467
|
+
const callUuid = endpoint.uuid;
|
|
468
|
+
|
|
469
|
+
console.log('[' + new Date().toISOString() + '] CALL Connected: ' + callUuid);
|
|
470
|
+
|
|
471
|
+
dialog.on('destroy', function() {
|
|
472
|
+
console.log('[' + new Date().toISOString() + '] CALL Ended');
|
|
473
|
+
if (endpoint) endpoint.destroy().catch(function() {});
|
|
474
|
+
});
|
|
475
|
+
|
|
476
|
+
await conversationLoop(endpoint, dialog, callUuid, options, deviceConfig);
|
|
477
|
+
return { endpoint: endpoint, dialog: dialog, callerId: callerId, callUuid: callUuid };
|
|
478
|
+
|
|
479
|
+
} catch (error) {
|
|
480
|
+
console.error('[' + new Date().toISOString() + '] CALL Error:', error.message);
|
|
481
|
+
try { res.send(500); } catch (e) {}
|
|
482
|
+
throw error;
|
|
483
|
+
}
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
module.exports = {
|
|
487
|
+
handleInvite: handleInvite,
|
|
488
|
+
extractCallerId: extractCallerId,
|
|
489
|
+
extractDialedExtension: extractDialedExtension
|
|
490
|
+
};
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text-to-Speech Service
|
|
3
|
+
*
|
|
4
|
+
* Default mode ("local"): sends text to the local Piper sidecar container
|
|
5
|
+
* (tts-local) and saves the returned WAV — no API key, fully offline.
|
|
6
|
+
*
|
|
7
|
+
* Optional mode ("cloud", set TTS_MODE=cloud): original ElevenLabs API
|
|
8
|
+
* behavior, preserved for anyone who still wants it.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
const axios = require('axios');
|
|
12
|
+
const fs = require('fs');
|
|
13
|
+
const path = require('path');
|
|
14
|
+
const crypto = require('crypto');
|
|
15
|
+
const logger = require('./logger');
|
|
16
|
+
|
|
17
|
+
const TTS_MODE = (process.env.TTS_MODE || 'local').toLowerCase();
|
|
18
|
+
const TTS_LOCAL_URL = process.env.TTS_LOCAL_URL || 'http://tts-local:9002';
|
|
19
|
+
const PIPER_VOICE = process.env.PIPER_VOICE || 'en_US-lessac-medium';
|
|
20
|
+
// Base URL other containers (FreeSWITCH) use to fetch generated audio back
|
|
21
|
+
// from this voice-app container. Under Docker Desktop bridge networking
|
|
22
|
+
// this must be the service name, not 127.0.0.1.
|
|
23
|
+
const AUDIO_BASE_URL = process.env.AUDIO_BASE_URL || 'http://voice-app:3000';
|
|
24
|
+
|
|
25
|
+
const ELEVENLABS_API_KEY = process.env.ELEVENLABS_API_KEY;
|
|
26
|
+
const ELEVENLABS_API_URL = 'https://api.elevenlabs.io/v1';
|
|
27
|
+
const DEFAULT_VOICE_ID = 'JAgnJveGGUh4qy4kh6dF';
|
|
28
|
+
const MODEL_ID = 'eleven_turbo_v2';
|
|
29
|
+
|
|
30
|
+
// Audio output directory (set via setAudioDir)
|
|
31
|
+
let audioDir = path.join(__dirname, '../audio-temp');
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Set the audio output directory
|
|
35
|
+
* @param {string} dir - Absolute path to audio directory
|
|
36
|
+
*/
|
|
37
|
+
function setAudioDir(dir) {
|
|
38
|
+
audioDir = dir;
|
|
39
|
+
if (!fs.existsSync(audioDir)) {
|
|
40
|
+
fs.mkdirSync(audioDir, { recursive: true });
|
|
41
|
+
logger.info('Created audio directory', { path: audioDir });
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function generateFilename(text, ext) {
|
|
46
|
+
const hash = crypto.createHash('md5').update(text).digest('hex').substring(0, 8);
|
|
47
|
+
const timestamp = Date.now();
|
|
48
|
+
return `tts-${timestamp}-${hash}.${ext}`;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Generate speech using the local Piper sidecar
|
|
53
|
+
* @param {string} text
|
|
54
|
+
* @param {string} voice - Piper voice name (falls back to PIPER_VOICE env)
|
|
55
|
+
* @returns {Promise<string>} HTTP URL to the generated WAV file
|
|
56
|
+
*/
|
|
57
|
+
async function generateSpeechLocal(text, voice) {
|
|
58
|
+
const startTime = Date.now();
|
|
59
|
+
|
|
60
|
+
logger.info('Generating speech with local Piper', { textLength: text.length, voice });
|
|
61
|
+
|
|
62
|
+
const response = await axios({
|
|
63
|
+
method: 'POST',
|
|
64
|
+
url: `${TTS_LOCAL_URL}/speak`,
|
|
65
|
+
headers: { 'Content-Type': 'application/json' },
|
|
66
|
+
data: { text, voice: voice || PIPER_VOICE },
|
|
67
|
+
responseType: 'arraybuffer',
|
|
68
|
+
timeout: 30000
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
const filename = generateFilename(text, 'wav');
|
|
72
|
+
const filepath = path.join(audioDir, filename);
|
|
73
|
+
fs.writeFileSync(filepath, response.data);
|
|
74
|
+
|
|
75
|
+
const latency = Date.now() - startTime;
|
|
76
|
+
logger.info('Speech generation successful (local)', {
|
|
77
|
+
filename, fileSize: response.data.length, latency, textLength: text.length
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
return `${AUDIO_BASE_URL}/audio-files/${filename}`;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Generate speech using ElevenLabs (cloud mode)
|
|
85
|
+
*/
|
|
86
|
+
async function generateSpeechCloud(text, voiceId = DEFAULT_VOICE_ID) {
|
|
87
|
+
const startTime = Date.now();
|
|
88
|
+
|
|
89
|
+
try {
|
|
90
|
+
if (!ELEVENLABS_API_KEY) {
|
|
91
|
+
throw new Error('ELEVENLABS_API_KEY environment variable not set');
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
logger.info('Generating speech with ElevenLabs', { textLength: text.length, voiceId, model: MODEL_ID });
|
|
95
|
+
|
|
96
|
+
const response = await axios({
|
|
97
|
+
method: 'POST',
|
|
98
|
+
url: `${ELEVENLABS_API_URL}/text-to-speech/${voiceId}`,
|
|
99
|
+
headers: {
|
|
100
|
+
'Accept': 'audio/mpeg',
|
|
101
|
+
'Content-Type': 'application/json',
|
|
102
|
+
'xi-api-key': ELEVENLABS_API_KEY
|
|
103
|
+
},
|
|
104
|
+
data: {
|
|
105
|
+
text,
|
|
106
|
+
model_id: MODEL_ID,
|
|
107
|
+
voice_settings: { stability: 0.5, similarity_boost: 0.75, style: 0.0, use_speaker_boost: true }
|
|
108
|
+
},
|
|
109
|
+
responseType: 'arraybuffer'
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
const filename = generateFilename(text, 'mp3');
|
|
113
|
+
const filepath = path.join(audioDir, filename);
|
|
114
|
+
fs.writeFileSync(filepath, response.data);
|
|
115
|
+
|
|
116
|
+
const latency = Date.now() - startTime;
|
|
117
|
+
logger.info('Speech generation successful (cloud)', {
|
|
118
|
+
filename, fileSize: response.data.length, latency, textLength: text.length
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
return `${AUDIO_BASE_URL}/audio-files/${filename}`;
|
|
122
|
+
} catch (error) {
|
|
123
|
+
const latency = Date.now() - startTime;
|
|
124
|
+
logger.error('Speech generation failed', {
|
|
125
|
+
error: error.message, latency, textLength: text?.length,
|
|
126
|
+
responseStatus: error.response?.status, responseData: error.response?.data?.toString()
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
if (error.response?.status === 401) throw new Error('ElevenLabs API authentication failed - check API key');
|
|
130
|
+
if (error.response?.status === 429) throw new Error('ElevenLabs API rate limit exceeded');
|
|
131
|
+
if (error.response?.status === 400) throw new Error('Invalid request to ElevenLabs API');
|
|
132
|
+
throw new Error(`TTS generation failed: ${error.message}`);
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Convert text to speech (local Piper by default, ElevenLabs if TTS_MODE=cloud)
|
|
138
|
+
* @param {string} text
|
|
139
|
+
* @param {string} voiceId - Piper voice name (local) or ElevenLabs voice ID (cloud)
|
|
140
|
+
* @returns {Promise<string>} HTTP URL to audio file
|
|
141
|
+
*/
|
|
142
|
+
async function generateSpeech(text, voiceId) {
|
|
143
|
+
if (TTS_MODE === 'cloud') {
|
|
144
|
+
return generateSpeechCloud(text, voiceId || DEFAULT_VOICE_ID);
|
|
145
|
+
}
|
|
146
|
+
return generateSpeechLocal(text, voiceId);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Clean up old audio files (older than specified age)
|
|
151
|
+
*/
|
|
152
|
+
function cleanupOldFiles(maxAgeMs = 60 * 60 * 1000) {
|
|
153
|
+
try {
|
|
154
|
+
const now = Date.now();
|
|
155
|
+
const files = fs.readdirSync(audioDir);
|
|
156
|
+
let deletedCount = 0;
|
|
157
|
+
files.forEach(file => {
|
|
158
|
+
if (!file.startsWith('tts-') || !(file.endsWith('.mp3') || file.endsWith('.wav'))) return;
|
|
159
|
+
const filepath = path.join(audioDir, file);
|
|
160
|
+
const stats = fs.statSync(filepath);
|
|
161
|
+
if (now - stats.mtimeMs > maxAgeMs) {
|
|
162
|
+
fs.unlinkSync(filepath);
|
|
163
|
+
deletedCount++;
|
|
164
|
+
}
|
|
165
|
+
});
|
|
166
|
+
if (deletedCount > 0) logger.info('Cleaned up old audio files', { deletedCount });
|
|
167
|
+
} catch (error) {
|
|
168
|
+
logger.warn('Failed to cleanup old audio files', { error: error.message });
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Get list of available voices
|
|
174
|
+
* @returns {Promise<Array>} local: Piper voices installed; cloud: ElevenLabs voices
|
|
175
|
+
*/
|
|
176
|
+
async function getAvailableVoices() {
|
|
177
|
+
if (TTS_MODE !== 'cloud') {
|
|
178
|
+
const response = await axios.get(`${TTS_LOCAL_URL}/voices`, { timeout: 5000 });
|
|
179
|
+
return response.data.voices || [];
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
if (!ELEVENLABS_API_KEY) {
|
|
183
|
+
throw new Error('ELEVENLABS_API_KEY environment variable not set');
|
|
184
|
+
}
|
|
185
|
+
const response = await axios({
|
|
186
|
+
method: 'GET',
|
|
187
|
+
url: `${ELEVENLABS_API_URL}/voices`,
|
|
188
|
+
headers: { 'xi-api-key': ELEVENLABS_API_KEY }
|
|
189
|
+
});
|
|
190
|
+
return response.data.voices;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// Initialize audio directory
|
|
194
|
+
setAudioDir(audioDir);
|
|
195
|
+
|
|
196
|
+
// Periodic cleanup (every 30 minutes)
|
|
197
|
+
setInterval(() => cleanupOldFiles(), 30 * 60 * 1000);
|
|
198
|
+
|
|
199
|
+
module.exports = {
|
|
200
|
+
generateSpeech,
|
|
201
|
+
setAudioDir,
|
|
202
|
+
cleanupOldFiles,
|
|
203
|
+
getAvailableVoices,
|
|
204
|
+
mode: TTS_MODE
|
|
205
|
+
};
|