claude-phone-local 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/.env.example +108 -0
  2. package/Dockerfile +84 -0
  3. package/LICENSE +22 -0
  4. package/README.md +231 -0
  5. package/claude-api-server/package-lock.json +832 -0
  6. package/claude-api-server/package.json +20 -0
  7. package/claude-api-server/server.js +540 -0
  8. package/claude-api-server/structured.js +208 -0
  9. package/cli/README.md +231 -0
  10. package/cli/bin/check-publish-safety.js +53 -0
  11. package/cli/bin/claude-phone.js +23 -0
  12. package/cli/bin/cli-main.js +284 -0
  13. package/cli/bin/postinstall.js +20 -0
  14. package/cli/lib/commands/api-server.js +77 -0
  15. package/cli/lib/commands/backup.js +69 -0
  16. package/cli/lib/commands/config/path.js +30 -0
  17. package/cli/lib/commands/config/reset.js +73 -0
  18. package/cli/lib/commands/config/show.js +86 -0
  19. package/cli/lib/commands/device/add.js +143 -0
  20. package/cli/lib/commands/device/list.js +55 -0
  21. package/cli/lib/commands/device/remove.js +83 -0
  22. package/cli/lib/commands/doctor.js +422 -0
  23. package/cli/lib/commands/logs.js +186 -0
  24. package/cli/lib/commands/restore.js +172 -0
  25. package/cli/lib/commands/setup.js +1246 -0
  26. package/cli/lib/commands/start.js +346 -0
  27. package/cli/lib/commands/status.js +139 -0
  28. package/cli/lib/commands/stop.js +101 -0
  29. package/cli/lib/commands/uninstall.js +183 -0
  30. package/cli/lib/commands/update.js +205 -0
  31. package/cli/lib/config.js +113 -0
  32. package/cli/lib/docker.js +384 -0
  33. package/cli/lib/mcp-register.js +108 -0
  34. package/cli/lib/network.js +109 -0
  35. package/cli/lib/platform.js +85 -0
  36. package/cli/lib/port-check.js +135 -0
  37. package/cli/lib/prereqs/checks/compose.js +147 -0
  38. package/cli/lib/prereqs/checks/disk.js +89 -0
  39. package/cli/lib/prereqs/checks/docker.js +155 -0
  40. package/cli/lib/prereqs/checks/network.js +78 -0
  41. package/cli/lib/prereqs/checks/node.js +95 -0
  42. package/cli/lib/prereqs/installers/docker-desktop.js +248 -0
  43. package/cli/lib/prereqs/installers/docker.js +229 -0
  44. package/cli/lib/prereqs/installers/node.js +254 -0
  45. package/cli/lib/prereqs/platform.js +175 -0
  46. package/cli/lib/prereqs/utils/execute.js +208 -0
  47. package/cli/lib/prereqs/utils/rollback.js +223 -0
  48. package/cli/lib/prereqs/utils/sudo.js +177 -0
  49. package/cli/lib/prereqs.js +228 -0
  50. package/cli/lib/prerequisites.js +115 -0
  51. package/cli/lib/process-manager.js +176 -0
  52. package/cli/lib/utils.js +78 -0
  53. package/cli/lib/validators.js +219 -0
  54. package/cli/lib/voice-downloader.js +79 -0
  55. package/docker/entrypoint.sh +110 -0
  56. package/docker/supervisord.conf +78 -0
  57. package/docker-compose.yml +40 -0
  58. package/docs/CLAUDE-CODE-SKILL.md +45 -0
  59. package/docs/LANGUAGES.md +113 -0
  60. package/docs/MCP-SERVER.md +111 -0
  61. package/docs/TROUBLESHOOTING.md +288 -0
  62. package/mcp-server/index.js +159 -0
  63. package/mcp-server/package-lock.json +1201 -0
  64. package/mcp-server/package.json +9 -0
  65. package/package.json +73 -0
  66. package/stt-local/Dockerfile +12 -0
  67. package/stt-local/server.py +45 -0
  68. package/tts-local/Dockerfile +9 -0
  69. package/tts-local/server.py +83 -0
  70. package/voice-app/API-QUERY-CONTRACT.md +238 -0
  71. package/voice-app/DEPLOYMENT.md +176 -0
  72. package/voice-app/Dockerfile +20 -0
  73. package/voice-app/README-OUTBOUND.md +316 -0
  74. package/voice-app/config/devices.json.example +18 -0
  75. package/voice-app/index.js +273 -0
  76. package/voice-app/lib/audio-fork.js +465 -0
  77. package/voice-app/lib/claude-bridge.js +113 -0
  78. package/voice-app/lib/connection-retry.js +58 -0
  79. package/voice-app/lib/conversation-loop.js +490 -0
  80. package/voice-app/lib/device-registry.js +171 -0
  81. package/voice-app/lib/http-server.js +242 -0
  82. package/voice-app/lib/logger.js +35 -0
  83. package/voice-app/lib/multi-registrar.js +133 -0
  84. package/voice-app/lib/outbound-handler.js +299 -0
  85. package/voice-app/lib/outbound-routes.js +417 -0
  86. package/voice-app/lib/outbound-session.js +324 -0
  87. package/voice-app/lib/query-routes.js +484 -0
  88. package/voice-app/lib/registrar.js +144 -0
  89. package/voice-app/lib/sip-handler.js +490 -0
  90. package/voice-app/lib/tts-service.js +205 -0
  91. package/voice-app/lib/whisper-client.js +140 -0
  92. package/voice-app/package.json +38 -0
  93. package/voice-app/static/gotit-beep.wav +0 -0
  94. package/voice-app/static/hold-music.wav +0 -0
  95. package/voice-app/static/ready-beep.wav +0 -0
  96. package/voice-app/test/freeswitch-retry.test.js +100 -0
@@ -0,0 +1,490 @@
1
+ /**
2
+ * SIP Call Handler with Conversation Loop
3
+ * v12: Device registry integration with proper method names
4
+ */
5
+
6
+ const { setTimeout: sleep } = require('node:timers/promises');
7
+
8
+ // FreeSWITCH (a separate container) fetches/connects to these - must be the
9
+ // voice-app container's address, not FreeSWITCH's own loopback.
10
+ const AUDIO_BASE_URL = process.env.AUDIO_BASE_URL || 'http://voice-app:3000';
11
+ const AUDIO_WS_HOST = new URL(AUDIO_BASE_URL).hostname;
12
+
13
+ // Audio cue URLs
14
+ const READY_BEEP_URL = `${AUDIO_BASE_URL}/static/ready-beep.wav`;
15
+ const GOTIT_BEEP_URL = `${AUDIO_BASE_URL}/static/gotit-beep.wav`;
16
+ const HOLD_MUSIC_URL = `${AUDIO_BASE_URL}/static/hold-music.wav`;
17
+
18
+ // Default voice ID (Morpheus)
19
+ const DEFAULT_VOICE_ID = 'JAgnJveGGUh4qy4kh6dF';
20
+
21
+ // Whisper language code -> Piper voice installed in tts-local/voices/.
22
+ // Whisper detects the caller's language per utterance and we answer in the
23
+ // same one. Override/extend with LANG_VOICE_MAP in .env as JSON.
24
+ const DEFAULT_LANG_VOICES = {
25
+ en: 'en_US-lessac-medium',
26
+ hi: 'hi_IN-priyamvada-medium',
27
+ mr: 'mr_IN-google-medium'
28
+ };
29
+
30
+ let LANG_VOICES = DEFAULT_LANG_VOICES;
31
+ if (process.env.LANG_VOICE_MAP) {
32
+ try {
33
+ LANG_VOICES = Object.assign({}, DEFAULT_LANG_VOICES, JSON.parse(process.env.LANG_VOICE_MAP));
34
+ } catch (e) {
35
+ console.log('[' + new Date().toISOString() + '] LANG: bad LANG_VOICE_MAP JSON, using defaults');
36
+ }
37
+ }
38
+
39
+ // Languages we will actually answer in. Anything Whisper detects outside this
40
+ // set falls back to the device's own voice, so a misdetection can't leave us
41
+ // with no installed model.
42
+ const SUPPORTED_LANGS = (process.env.SUPPORTED_LANGS || 'en,hi,mr')
43
+ .split(',').map(function (x) { return x.trim(); }).filter(Boolean);
44
+
45
+ function voiceForLanguage(lang, fallbackVoice) {
46
+ if (!lang) return fallbackVoice;
47
+ if (SUPPORTED_LANGS.indexOf(lang) === -1) return fallbackVoice;
48
+ return LANG_VOICES[lang] || fallbackVoice;
49
+ }
50
+
51
+ // Claude Code-style thinking phrases
52
+ const THINKING_PHRASES = [
53
+ "Let me check that for you.",
54
+ "One moment.",
55
+ "Just a second.",
56
+ "Looking into it now.",
57
+ "Give me a moment.",
58
+ "Checking on that.",
59
+ "Hang on, almost there.",
60
+ "Still working on it.",
61
+ "Nearly done.",
62
+ "Bear with me a moment.",
63
+ ];
64
+
65
+ // Said only while the caller is already waiting, so they never hear the same
66
+ // line twice in a row within one wait.
67
+ const WAITING_PHRASES = [
68
+ "Still working on this.",
69
+ "Almost there.",
70
+ "Just a little longer.",
71
+ "Nearly finished.",
72
+ "Hang in there, still going.",
73
+ "Won't be much longer.",
74
+ ];
75
+
76
+ function getRandomThinkingPhrase() {
77
+ return THINKING_PHRASES[Math.floor(Math.random() * THINKING_PHRASES.length)];
78
+ }
79
+
80
+ function extractCallerId(req) {
81
+ var from = req.get("From") || "";
82
+ var match = from.match(/sip:([+\d]+)@/);
83
+ if (match) return match[1];
84
+ var numMatch = from.match(/<sip:(\d+)@/);
85
+ if (numMatch) return numMatch[1];
86
+ return "unknown";
87
+ }
88
+
89
+ /**
90
+ * Extract dialed extension from SIP To header
91
+ */
92
+ function extractDialedExtension(req) {
93
+ var to = req.get("To") || "";
94
+ var match = to.match(/sip:(\d+)@/);
95
+ if (match) {
96
+ return match[1];
97
+ }
98
+ return null;
99
+ }
100
+
101
+ function isGoodbye(transcript) {
102
+ const lower = transcript.toLowerCase().trim();
103
+ const goodbyePhrases = [
104
+ // English
105
+ 'goodbye', 'good bye', 'bye', 'bye bye', 'hang up', 'end call', 'end the call',
106
+ 'close the call', 'cut the call', 'disconnect', "that's all", 'thats all',
107
+ 'thank you bye', 'talk later',
108
+ // Hindi / Marathi (Devanagari + common romanisations)
109
+ 'अलविदा', 'नमस्ते', 'बाय', 'फोन बंद करो', 'कॉल बंद करो', 'बंद करो',
110
+ 'ठेवतो', 'ठेवते', 'फोन ठेव', 'बंद कर',
111
+ 'alvida', 'phone band karo', 'call band karo', 'band karo', 'thevto'
112
+ ];
113
+ return goodbyePhrases.some(function(phrase) {
114
+ return lower === phrase || lower.includes(' ' + phrase) ||
115
+ lower.startsWith(phrase + ' ') || lower.endsWith(' ' + phrase);
116
+ });
117
+ }
118
+
119
+ /**
120
+ * Extract voice-friendly line from Claude's response
121
+ * Priority: VOICE_RESPONSE > CUSTOM COMPLETED > COMPLETED > first sentence
122
+ */
123
+ function extractVoiceLine(response) {
124
+ // Priority 1: VOICE_RESPONSE (new format)
125
+ var voiceMatch = response.match(/🗣️\s*VOICE_RESPONSE:\s*([^\n]+)/im);
126
+ if (voiceMatch) {
127
+ var text = voiceMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
128
+ if (text && text.split(/\s+/).length <= 60) {
129
+ return text;
130
+ }
131
+ }
132
+
133
+ // Priority 2: CUSTOM COMPLETED
134
+ var customMatch = response.match(/🗣️\s*CUSTOM\s+COMPLETED:\s*(.+?)(?:\n|$)/im);
135
+ if (customMatch) {
136
+ text = customMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
137
+ if (text && text.split(/\s+/).length <= 50) {
138
+ return text;
139
+ }
140
+ }
141
+
142
+ // Priority 3: COMPLETED
143
+ var completedMatch = response.match(/🎯\s*COMPLETED:\s*(.+?)(?:\n|$)/im);
144
+ if (completedMatch) {
145
+ return completedMatch[1].trim().replace(/\*+/g, '').replace(/\[.*?\]/g, '').trim();
146
+ }
147
+
148
+ // Priority 4: First sentence
149
+ var firstSentence = response.split(/[.!?]/)[0];
150
+ if (firstSentence && firstSentence.length < 500) {
151
+ return firstSentence.trim();
152
+ }
153
+
154
+ return response.substring(0, 500).trim();
155
+ }
156
+
157
+ /**
158
+ * Play a clip the caller is allowed to interrupt.
159
+ *
160
+ * FreeSWITCH plays to completion unless told otherwise, so to support barge-in
161
+ * we arm the detector, then issue uuid_break the moment the caller starts
162
+ * talking. Returns true if the caller cut in, so the loop can skip straight to
163
+ * listening instead of finishing what it was saying.
164
+ */
165
+ async function playInterruptible(endpoint, session, url) {
166
+ if (!session) {
167
+ await endpoint.play(url);
168
+ return false;
169
+ }
170
+
171
+ let barged = false;
172
+ const onBarge = function () {
173
+ barged = true;
174
+ // uuid_break stops the current playback on this leg immediately.
175
+ endpoint.api('uuid_break', endpoint.uuid).catch(function () {});
176
+ };
177
+
178
+ session.once('barge-in', onBarge);
179
+ session.setBargeInEnabled(true);
180
+ try {
181
+ await endpoint.play(url);
182
+ } finally {
183
+ session.setBargeInEnabled(false);
184
+ session.removeListener('barge-in', onBarge);
185
+ }
186
+ if (barged) {
187
+ console.log('[' + new Date().toISOString() + '] BARGE-IN: caller interrupted, listening now');
188
+ }
189
+ return barged;
190
+ }
191
+
192
+ /**
193
+ * Main conversation loop
194
+ * @param {Object} deviceConfig - Device configuration (name, prompt, voiceId, etc.) or null for default
195
+ */
196
+ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfig) {
197
+ const { ttsService, whisperClient, claudeBridge, wsPort, audioForkServer } = options;
198
+
199
+ let session = null;
200
+ let forkRunning = false;
201
+
202
+ // Get device-specific settings
203
+ const deviceName = deviceConfig ? deviceConfig.name : 'Morpheus';
204
+ const devicePrompt = deviceConfig ? deviceConfig.prompt : null;
205
+ // Local mode: device.voice is a Piper voice name (null falls back to PIPER_VOICE env).
206
+ // Cloud mode: device.voiceId is an ElevenLabs voice ID (falls back to DEFAULT_VOICE_ID).
207
+ const voiceId = ttsService.mode === 'cloud'
208
+ ? ((deviceConfig && deviceConfig.voiceId) ? deviceConfig.voiceId : DEFAULT_VOICE_ID)
209
+ : ((deviceConfig && deviceConfig.voice) ? deviceConfig.voice : null);
210
+ // Voice used for the current turn - starts as the device voice and follows
211
+ // the caller's detected language from the first utterance onward.
212
+ let turnVoice = ttsService.mode === 'cloud'
213
+ ? (deviceConfig && deviceConfig.voiceId ? deviceConfig.voiceId : DEFAULT_VOICE_ID)
214
+ : (deviceConfig && deviceConfig.voice ? deviceConfig.voice : null);
215
+
216
+ const greeting = deviceConfig && deviceConfig.name !== 'Morpheus'
217
+ ? "Hello! I'm " + deviceConfig.name + ". How can I help you today?"
218
+ : "Hello! I'm your server. How can I help you today?";
219
+
220
+ try {
221
+ console.log('[' + new Date().toISOString() + '] CONVERSATION Starting (session: ' + callUuid + ', device: ' + deviceName + ', voice: ' + voiceId + ')...');
222
+
223
+ // Play device-specific greeting with device voice BEFORE starting audio fork
224
+ console.log('[' + new Date().toISOString() + '] Generating greeting...');
225
+ const greetingUrl = await ttsService.generateSpeech(greeting, voiceId);
226
+ console.log('[' + new Date().toISOString() + '] Playing greeting: ' + greetingUrl);
227
+ await endpoint.play(greetingUrl);
228
+ console.log('[' + new Date().toISOString() + '] Greeting played successfully');
229
+
230
+ // Start fork for entire call AFTER greeting
231
+ const wsUrl = 'ws://' + AUDIO_WS_HOST + ':' + wsPort + '/' + encodeURIComponent(callUuid);
232
+ const sessionPromise = audioForkServer.expectSession(callUuid, { timeoutMs: 10000 });
233
+
234
+ await endpoint.forkAudioStart({
235
+ wsUrl: wsUrl,
236
+ mixType: 'mono',
237
+ sampling: '16k'
238
+ });
239
+ forkRunning = true;
240
+
241
+ session = await sessionPromise;
242
+ console.log('[' + new Date().toISOString() + '] AUDIO Fork connected');
243
+
244
+ // Main conversation loop
245
+ let turnCount = 0;
246
+ const MAX_TURNS = 20;
247
+
248
+ while (turnCount < MAX_TURNS) {
249
+ turnCount++;
250
+ console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + '/' + MAX_TURNS);
251
+
252
+ // READY BEEP
253
+ try {
254
+ await endpoint.play(READY_BEEP_URL);
255
+ } catch (e) {
256
+ console.log('[' + new Date().toISOString() + '] BEEP: Ready beep failed, continuing');
257
+ }
258
+
259
+ session.setCaptureEnabled(true);
260
+ console.log('[' + new Date().toISOString() + '] LISTEN Waiting for speech...');
261
+
262
+ let utterance = null;
263
+ try {
264
+ utterance = await session.waitForUtterance({ timeoutMs: 30000 });
265
+ console.log('[' + new Date().toISOString() + '] LISTEN Got: ' + utterance.audio.length + ' bytes');
266
+ } catch (err) {
267
+ console.log('[' + new Date().toISOString() + '] LISTEN Timeout: ' + err.message);
268
+ }
269
+
270
+ session.setCaptureEnabled(false);
271
+
272
+ if (!utterance) {
273
+ const promptUrl = await ttsService.generateSpeech("I didn't hear anything. Are you still there?", turnVoice);
274
+ await endpoint.play(promptUrl);
275
+ continue;
276
+ }
277
+
278
+ // GOT-IT BEEP
279
+ try {
280
+ await endpoint.play(GOTIT_BEEP_URL);
281
+ } catch (e) {
282
+ console.log('[' + new Date().toISOString() + '] BEEP: Got-it beep failed, continuing');
283
+ }
284
+
285
+ // Transcribe (language auto-detected unless STT_LANGUAGE pins one)
286
+ const sttResult = await whisperClient.transcribeDetailed(utterance.audio, {
287
+ format: 'pcm',
288
+ sampleRate: 16000,
289
+ language: (deviceConfig && deviceConfig.language) || process.env.STT_LANGUAGE || 'auto'
290
+ });
291
+ const transcript = sttResult.text;
292
+ const detectedLang = sttResult.language;
293
+
294
+ // Answer in whatever language the caller just used.
295
+ turnVoice = voiceForLanguage(detectedLang, voiceId);
296
+ console.log('[' + new Date().toISOString() + '] WHISPER [' + (detectedLang || '?') +
297
+ ' -> voice ' + turnVoice + ']: "' + transcript + '"');
298
+
299
+ if (!transcript || transcript.trim().length < 2) {
300
+ const clarifyUrl = await ttsService.generateSpeech("Sorry, I didn't catch that. Could you repeat?", turnVoice);
301
+ await endpoint.play(clarifyUrl);
302
+ continue;
303
+ }
304
+
305
+ if (isGoodbye(transcript)) {
306
+ const byeUrl = await ttsService.generateSpeech("Goodbye! Call again anytime.", turnVoice);
307
+ await endpoint.play(byeUrl);
308
+ break;
309
+ }
310
+
311
+ // THINKING FEEDBACK
312
+ const thinkingPhrase = getRandomThinkingPhrase();
313
+ console.log('[' + new Date().toISOString() + '] THINKING: "' + thinkingPhrase + '"');
314
+ const thinkingUrl = await ttsService.generateSpeech(thinkingPhrase, turnVoice);
315
+ await endpoint.play(thinkingUrl);
316
+
317
+ // hears nothing for the whole query and hangs up.
318
+ // Fill the whole wait, not just parts of it. The gap alternates between a
319
+ // soft music bed and a spoken line, so the line never goes dead. Clips play
320
+ // to completion - one endpoint cannot layer two streams - and every Nth
321
+ // round is speech instead of music.
322
+ let waiting = true;
323
+ const SPEAK_EVERY = parseInt(process.env.KEEPALIVE_SPEAK_EVERY || '3', 10);
324
+ const keepAlive = (async function () {
325
+ let round = 0;
326
+ while (waiting) {
327
+ round++;
328
+ try {
329
+ const clipUrl = (round % SPEAK_EVERY === 0)
330
+ ? await ttsService.generateSpeech(getRandomWaitingPhrase(), turnVoice)
331
+ : HOLD_MUSIC_URL;
332
+ if (!waiting) break;
333
+ // Caller can cut through the hold music / filler to add something.
334
+ if (await playInterruptible(endpoint, session, clipUrl)) {
335
+ waiting = false;
336
+ break;
337
+ }
338
+ } catch (e) {
339
+ console.log('[' + new Date().toISOString() + '] KEEPALIVE: stopped (' + e.message + ')');
340
+ return;
341
+ }
342
+ }
343
+ })();
344
+
345
+ // Query Claude with device-specific prompt
346
+ console.log('[' + new Date().toISOString() + '] CLAUDE Querying (device: ' + deviceName + ')...');
347
+ let claudeResponse;
348
+ try {
349
+ claudeResponse = await claudeBridge.query(
350
+ transcript,
351
+ { callId: callUuid, devicePrompt: devicePrompt }
352
+ );
353
+ } finally {
354
+ waiting = false;
355
+ try { await keepAlive; } catch (e) {}
356
+ }
357
+
358
+ console.log('[' + new Date().toISOString() + '] CLAUDE Response received');
359
+
360
+ // Extract and play voice line with device voice
361
+ const voiceLine = extractVoiceLine(claudeResponse);
362
+ console.log('[' + new Date().toISOString() + '] VOICE: "' + voiceLine + '"');
363
+
364
+ const responseUrl = await ttsService.generateSpeech(voiceLine, turnVoice);
365
+ // Long answers are the usual thing people want to interrupt.
366
+ await playInterruptible(endpoint, session, responseUrl);
367
+
368
+ console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + ' complete');
369
+ }
370
+
371
+ if (turnCount >= MAX_TURNS) {
372
+ const maxUrl = await ttsService.generateSpeech("We've been talking for a while. Goodbye!", turnVoice);
373
+ await endpoint.play(maxUrl);
374
+ }
375
+
376
+ } catch (error) {
377
+ console.error('[' + new Date().toISOString() + '] CONVERSATION Error:', error.message);
378
+ try {
379
+ if (session) session.setCaptureEnabled(false);
380
+ const errUrl = await ttsService.generateSpeech("Sorry, something went wrong.", voiceId);
381
+ await endpoint.play(errUrl);
382
+ } catch (e) {}
383
+ } finally {
384
+ console.log('[' + new Date().toISOString() + '] CONVERSATION Cleanup...');
385
+
386
+ try {
387
+ await claudeBridge.endSession(callUuid);
388
+ } catch (e) {}
389
+
390
+ if (forkRunning) {
391
+ try {
392
+ await endpoint.forkAudioStop();
393
+ } catch (e) {}
394
+ }
395
+
396
+ try { dialog.destroy(); } catch (e) {}
397
+ }
398
+ }
399
+
400
+ /**
401
+ * Strip video tracks from SDP (FreeSWITCH doesn't support H.261 and rejects with 488)
402
+ * Keeps only audio tracks to ensure codec negotiation succeeds
403
+ */
404
+ function stripVideoFromSdp(sdp) {
405
+ if (!sdp) return sdp;
406
+
407
+ const lines = sdp.split('\r\n');
408
+ const result = [];
409
+ let inVideoSection = false;
410
+
411
+ for (const line of lines) {
412
+ // Check if we're entering a video media section
413
+ if (line.startsWith('m=video')) {
414
+ inVideoSection = true;
415
+ continue; // Skip the m=video line
416
+ }
417
+
418
+ // Check if we're entering a new media section (audio, etc.)
419
+ if (line.startsWith('m=') && !line.startsWith('m=video')) {
420
+ inVideoSection = false;
421
+ }
422
+
423
+ // Skip all lines in the video section
424
+ if (inVideoSection) {
425
+ continue;
426
+ }
427
+
428
+ result.push(line);
429
+ }
430
+
431
+ return result.join('\r\n');
432
+ }
433
+
434
+ /**
435
+ * Handle incoming SIP INVITE
436
+ */
437
+ async function handleInvite(req, res, options) {
438
+ const { mediaServer, deviceRegistry } = options;
439
+
440
+ const callerId = extractCallerId(req);
441
+ const dialedExt = extractDialedExtension(req);
442
+
443
+ // Look up device config using deviceRegistry.get() (works with name OR extension)
444
+ let deviceConfig = null;
445
+ if (deviceRegistry && dialedExt) {
446
+ deviceConfig = deviceRegistry.get(dialedExt);
447
+ if (deviceConfig) {
448
+ console.log('[' + new Date().toISOString() + '] CALL Device matched: ' + deviceConfig.name + ' (ext ' + dialedExt + ')');
449
+ } else {
450
+ console.log('[' + new Date().toISOString() + '] CALL Unknown extension ' + dialedExt + ', using default');
451
+ deviceConfig = deviceRegistry.getDefault();
452
+ }
453
+ }
454
+
455
+ console.log('[' + new Date().toISOString() + '] CALL Incoming from: ' + callerId + ' to ext: ' + (dialedExt || 'unknown'));
456
+
457
+ try {
458
+ // Strip video from SDP to avoid FreeSWITCH 488 error with unsupported video codecs
459
+ const originalSdp = req.body;
460
+ const audioOnlySdp = stripVideoFromSdp(originalSdp);
461
+ if (originalSdp !== audioOnlySdp) {
462
+ console.log('[' + new Date().toISOString() + '] CALL Stripped video track from SDP');
463
+ }
464
+
465
+ const result = await mediaServer.connectCaller(req, res, { remoteSdp: audioOnlySdp });
466
+ const { endpoint, dialog } = result;
467
+ const callUuid = endpoint.uuid;
468
+
469
+ console.log('[' + new Date().toISOString() + '] CALL Connected: ' + callUuid);
470
+
471
+ dialog.on('destroy', function() {
472
+ console.log('[' + new Date().toISOString() + '] CALL Ended');
473
+ if (endpoint) endpoint.destroy().catch(function() {});
474
+ });
475
+
476
+ await conversationLoop(endpoint, dialog, callUuid, options, deviceConfig);
477
+ return { endpoint: endpoint, dialog: dialog, callerId: callerId, callUuid: callUuid };
478
+
479
+ } catch (error) {
480
+ console.error('[' + new Date().toISOString() + '] CALL Error:', error.message);
481
+ try { res.send(500); } catch (e) {}
482
+ throw error;
483
+ }
484
+ }
485
+
486
+ module.exports = {
487
+ handleInvite: handleInvite,
488
+ extractCallerId: extractCallerId,
489
+ extractDialedExtension: extractDialedExtension
490
+ };
@@ -0,0 +1,205 @@
1
+ /**
2
+ * Text-to-Speech Service
3
+ *
4
+ * Default mode ("local"): sends text to the local Piper sidecar container
5
+ * (tts-local) and saves the returned WAV — no API key, fully offline.
6
+ *
7
+ * Optional mode ("cloud", set TTS_MODE=cloud): original ElevenLabs API
8
+ * behavior, preserved for anyone who still wants it.
9
+ */
10
+
11
+ const axios = require('axios');
12
+ const fs = require('fs');
13
+ const path = require('path');
14
+ const crypto = require('crypto');
15
+ const logger = require('./logger');
16
+
17
+ const TTS_MODE = (process.env.TTS_MODE || 'local').toLowerCase();
18
+ const TTS_LOCAL_URL = process.env.TTS_LOCAL_URL || 'http://tts-local:9002';
19
+ const PIPER_VOICE = process.env.PIPER_VOICE || 'en_US-lessac-medium';
20
+ // Base URL other containers (FreeSWITCH) use to fetch generated audio back
21
+ // from this voice-app container. Under Docker Desktop bridge networking
22
+ // this must be the service name, not 127.0.0.1.
23
+ const AUDIO_BASE_URL = process.env.AUDIO_BASE_URL || 'http://voice-app:3000';
24
+
25
+ const ELEVENLABS_API_KEY = process.env.ELEVENLABS_API_KEY;
26
+ const ELEVENLABS_API_URL = 'https://api.elevenlabs.io/v1';
27
+ const DEFAULT_VOICE_ID = 'JAgnJveGGUh4qy4kh6dF';
28
+ const MODEL_ID = 'eleven_turbo_v2';
29
+
30
+ // Audio output directory (set via setAudioDir)
31
+ let audioDir = path.join(__dirname, '../audio-temp');
32
+
33
+ /**
34
+ * Set the audio output directory
35
+ * @param {string} dir - Absolute path to audio directory
36
+ */
37
+ function setAudioDir(dir) {
38
+ audioDir = dir;
39
+ if (!fs.existsSync(audioDir)) {
40
+ fs.mkdirSync(audioDir, { recursive: true });
41
+ logger.info('Created audio directory', { path: audioDir });
42
+ }
43
+ }
44
+
45
+ function generateFilename(text, ext) {
46
+ const hash = crypto.createHash('md5').update(text).digest('hex').substring(0, 8);
47
+ const timestamp = Date.now();
48
+ return `tts-${timestamp}-${hash}.${ext}`;
49
+ }
50
+
51
+ /**
52
+ * Generate speech using the local Piper sidecar
53
+ * @param {string} text
54
+ * @param {string} voice - Piper voice name (falls back to PIPER_VOICE env)
55
+ * @returns {Promise<string>} HTTP URL to the generated WAV file
56
+ */
57
+ async function generateSpeechLocal(text, voice) {
58
+ const startTime = Date.now();
59
+
60
+ logger.info('Generating speech with local Piper', { textLength: text.length, voice });
61
+
62
+ const response = await axios({
63
+ method: 'POST',
64
+ url: `${TTS_LOCAL_URL}/speak`,
65
+ headers: { 'Content-Type': 'application/json' },
66
+ data: { text, voice: voice || PIPER_VOICE },
67
+ responseType: 'arraybuffer',
68
+ timeout: 30000
69
+ });
70
+
71
+ const filename = generateFilename(text, 'wav');
72
+ const filepath = path.join(audioDir, filename);
73
+ fs.writeFileSync(filepath, response.data);
74
+
75
+ const latency = Date.now() - startTime;
76
+ logger.info('Speech generation successful (local)', {
77
+ filename, fileSize: response.data.length, latency, textLength: text.length
78
+ });
79
+
80
+ return `${AUDIO_BASE_URL}/audio-files/${filename}`;
81
+ }
82
+
83
+ /**
84
+ * Generate speech using ElevenLabs (cloud mode)
85
+ */
86
+ async function generateSpeechCloud(text, voiceId = DEFAULT_VOICE_ID) {
87
+ const startTime = Date.now();
88
+
89
+ try {
90
+ if (!ELEVENLABS_API_KEY) {
91
+ throw new Error('ELEVENLABS_API_KEY environment variable not set');
92
+ }
93
+
94
+ logger.info('Generating speech with ElevenLabs', { textLength: text.length, voiceId, model: MODEL_ID });
95
+
96
+ const response = await axios({
97
+ method: 'POST',
98
+ url: `${ELEVENLABS_API_URL}/text-to-speech/${voiceId}`,
99
+ headers: {
100
+ 'Accept': 'audio/mpeg',
101
+ 'Content-Type': 'application/json',
102
+ 'xi-api-key': ELEVENLABS_API_KEY
103
+ },
104
+ data: {
105
+ text,
106
+ model_id: MODEL_ID,
107
+ voice_settings: { stability: 0.5, similarity_boost: 0.75, style: 0.0, use_speaker_boost: true }
108
+ },
109
+ responseType: 'arraybuffer'
110
+ });
111
+
112
+ const filename = generateFilename(text, 'mp3');
113
+ const filepath = path.join(audioDir, filename);
114
+ fs.writeFileSync(filepath, response.data);
115
+
116
+ const latency = Date.now() - startTime;
117
+ logger.info('Speech generation successful (cloud)', {
118
+ filename, fileSize: response.data.length, latency, textLength: text.length
119
+ });
120
+
121
+ return `${AUDIO_BASE_URL}/audio-files/${filename}`;
122
+ } catch (error) {
123
+ const latency = Date.now() - startTime;
124
+ logger.error('Speech generation failed', {
125
+ error: error.message, latency, textLength: text?.length,
126
+ responseStatus: error.response?.status, responseData: error.response?.data?.toString()
127
+ });
128
+
129
+ if (error.response?.status === 401) throw new Error('ElevenLabs API authentication failed - check API key');
130
+ if (error.response?.status === 429) throw new Error('ElevenLabs API rate limit exceeded');
131
+ if (error.response?.status === 400) throw new Error('Invalid request to ElevenLabs API');
132
+ throw new Error(`TTS generation failed: ${error.message}`);
133
+ }
134
+ }
135
+
136
+ /**
137
+ * Convert text to speech (local Piper by default, ElevenLabs if TTS_MODE=cloud)
138
+ * @param {string} text
139
+ * @param {string} voiceId - Piper voice name (local) or ElevenLabs voice ID (cloud)
140
+ * @returns {Promise<string>} HTTP URL to audio file
141
+ */
142
+ async function generateSpeech(text, voiceId) {
143
+ if (TTS_MODE === 'cloud') {
144
+ return generateSpeechCloud(text, voiceId || DEFAULT_VOICE_ID);
145
+ }
146
+ return generateSpeechLocal(text, voiceId);
147
+ }
148
+
149
+ /**
150
+ * Clean up old audio files (older than specified age)
151
+ */
152
+ function cleanupOldFiles(maxAgeMs = 60 * 60 * 1000) {
153
+ try {
154
+ const now = Date.now();
155
+ const files = fs.readdirSync(audioDir);
156
+ let deletedCount = 0;
157
+ files.forEach(file => {
158
+ if (!file.startsWith('tts-') || !(file.endsWith('.mp3') || file.endsWith('.wav'))) return;
159
+ const filepath = path.join(audioDir, file);
160
+ const stats = fs.statSync(filepath);
161
+ if (now - stats.mtimeMs > maxAgeMs) {
162
+ fs.unlinkSync(filepath);
163
+ deletedCount++;
164
+ }
165
+ });
166
+ if (deletedCount > 0) logger.info('Cleaned up old audio files', { deletedCount });
167
+ } catch (error) {
168
+ logger.warn('Failed to cleanup old audio files', { error: error.message });
169
+ }
170
+ }
171
+
172
+ /**
173
+ * Get list of available voices
174
+ * @returns {Promise<Array>} local: Piper voices installed; cloud: ElevenLabs voices
175
+ */
176
+ async function getAvailableVoices() {
177
+ if (TTS_MODE !== 'cloud') {
178
+ const response = await axios.get(`${TTS_LOCAL_URL}/voices`, { timeout: 5000 });
179
+ return response.data.voices || [];
180
+ }
181
+
182
+ if (!ELEVENLABS_API_KEY) {
183
+ throw new Error('ELEVENLABS_API_KEY environment variable not set');
184
+ }
185
+ const response = await axios({
186
+ method: 'GET',
187
+ url: `${ELEVENLABS_API_URL}/voices`,
188
+ headers: { 'xi-api-key': ELEVENLABS_API_KEY }
189
+ });
190
+ return response.data.voices;
191
+ }
192
+
193
+ // Initialize audio directory
194
+ setAudioDir(audioDir);
195
+
196
+ // Periodic cleanup (every 30 minutes)
197
+ setInterval(() => cleanupOldFiles(), 30 * 60 * 1000);
198
+
199
+ module.exports = {
200
+ generateSpeech,
201
+ setAudioDir,
202
+ cleanupOldFiles,
203
+ getAvailableVoices,
204
+ mode: TTS_MODE
205
+ };