claude-phone-local 2.0.12 → 2.0.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "claude-phone-local",
|
|
3
|
-
"version": "2.0.
|
|
3
|
+
"version": "2.0.13",
|
|
4
4
|
"description": "Local/offline fork of NetworkChuck's claude-phone: talk to Claude Code over 3CX/SIP with faster-whisper STT + Piper TTS in one Docker container.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -278,6 +278,14 @@ class AudioForkSession extends EventEmitter {
|
|
|
278
278
|
this.bargeInEnabled = false;
|
|
279
279
|
console.log('[AUDIO-DEBUG] BARGE-IN detected (RMS=' + Math.round(stats.rms) + ') for ' + this.callUuid);
|
|
280
280
|
this.emit('barge-in');
|
|
281
|
+
// Barge-in used to only stop playback - the words that triggered it,
|
|
282
|
+
// and everything the caller kept saying until the turn loop's next
|
|
283
|
+
// setCaptureEnabled(true), were silently discarded because capture
|
|
284
|
+
// stayed off the whole time. Turn capture on immediately so this
|
|
285
|
+
// chunk (already loud speech, just confirmed above) and what follows
|
|
286
|
+
// fall through into real utterance capture below instead of being
|
|
287
|
+
// dropped and forcing the caller to repeat themselves.
|
|
288
|
+
this.captureEnabled = true;
|
|
281
289
|
}
|
|
282
290
|
}
|
|
283
291
|
|
|
@@ -163,13 +163,20 @@ function extractVoiceLine(response) {
|
|
|
163
163
|
*
|
|
164
164
|
* FreeSWITCH plays to completion unless told otherwise, so to support barge-in
|
|
165
165
|
* we arm the detector, then issue uuid_break the moment the caller starts
|
|
166
|
-
* talking.
|
|
167
|
-
*
|
|
166
|
+
* talking. AudioForkSession turns capture on itself right when barge-in
|
|
167
|
+
* fires (see audio-fork.js), so the words that triggered the interruption are
|
|
168
|
+
* already becoming a real utterance - previously they were discarded (capture
|
|
169
|
+
* stayed off during playback) and the caller had to repeat themselves after a
|
|
170
|
+
* fresh ready-beep on the next turn.
|
|
171
|
+
*
|
|
172
|
+
* Returns the captured utterance if the caller interrupted (so the turn loop
|
|
173
|
+
* can use it as their next input directly, no beep/re-prompt needed), or
|
|
174
|
+
* `null` if playback completed without interruption.
|
|
168
175
|
*/
|
|
169
176
|
async function playInterruptible(endpoint, session, url) {
|
|
170
177
|
if (!session) {
|
|
171
178
|
await endpoint.play(url);
|
|
172
|
-
return
|
|
179
|
+
return null;
|
|
173
180
|
}
|
|
174
181
|
|
|
175
182
|
let barged = false;
|
|
@@ -187,10 +194,25 @@ async function playInterruptible(endpoint, session, url) {
|
|
|
187
194
|
session.setBargeInEnabled(false);
|
|
188
195
|
session.removeListener('barge-in', onBarge);
|
|
189
196
|
}
|
|
190
|
-
|
|
191
|
-
|
|
197
|
+
|
|
198
|
+
if (!barged) return null;
|
|
199
|
+
|
|
200
|
+
console.log('[' + new Date().toISOString() + '] BARGE-IN: caller interrupted, capturing what they said');
|
|
201
|
+
try {
|
|
202
|
+
// Capture already started the moment barge-in fired; this just waits for
|
|
203
|
+
// the utterance to finish (end-of-speech silence) rather than starting a
|
|
204
|
+
// fresh listen window that would miss the words already spoken.
|
|
205
|
+
const utterance = await session.waitForUtterance({ timeoutMs: 15000 });
|
|
206
|
+
// finalizeUtterance() resets utterance state but not captureEnabled -
|
|
207
|
+
// turn it off now, otherwise every chunk while we transcribe/query
|
|
208
|
+
// Claude keeps accumulating into a new stray utterance.
|
|
209
|
+
session.setCaptureEnabled(false);
|
|
210
|
+
return utterance;
|
|
211
|
+
} catch (err) {
|
|
212
|
+
console.log('[' + new Date().toISOString() + '] BARGE-IN: capture failed (' + err.message + '), falling back to a fresh listen');
|
|
213
|
+
session.setCaptureEnabled(false);
|
|
214
|
+
return null;
|
|
192
215
|
}
|
|
193
|
-
return barged;
|
|
194
216
|
}
|
|
195
217
|
|
|
196
218
|
/**
|
|
@@ -253,30 +275,42 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
|
|
|
253
275
|
// Main conversation loop
|
|
254
276
|
let turnCount = 0;
|
|
255
277
|
const MAX_TURNS = 20;
|
|
278
|
+
// Speech captured by a barge-in during the previous turn's playback -
|
|
279
|
+
// consumed as this turn's input directly instead of a fresh ready-beep
|
|
280
|
+
// and listen window, which would otherwise make the caller repeat
|
|
281
|
+
// themselves after every interruption.
|
|
282
|
+
let pendingUtterance = null;
|
|
256
283
|
|
|
257
284
|
while (turnCount < MAX_TURNS) {
|
|
258
285
|
turnCount++;
|
|
259
286
|
console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + '/' + MAX_TURNS);
|
|
260
287
|
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
288
|
+
let utterance = null;
|
|
289
|
+
|
|
290
|
+
if (pendingUtterance) {
|
|
291
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Using speech captured during barge-in: ' + pendingUtterance.audio.length + ' bytes');
|
|
292
|
+
utterance = pendingUtterance;
|
|
293
|
+
pendingUtterance = null;
|
|
294
|
+
} else {
|
|
295
|
+
// READY BEEP
|
|
296
|
+
try {
|
|
297
|
+
await endpoint.play(READY_BEEP_URL);
|
|
298
|
+
} catch (e) {
|
|
299
|
+
console.log('[' + new Date().toISOString() + '] BEEP: Ready beep failed, continuing');
|
|
300
|
+
}
|
|
267
301
|
|
|
268
|
-
|
|
269
|
-
|
|
302
|
+
session.setCaptureEnabled(true);
|
|
303
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Waiting for speech...');
|
|
270
304
|
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
}
|
|
305
|
+
try {
|
|
306
|
+
utterance = await session.waitForUtterance({ timeoutMs: 30000 });
|
|
307
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Got: ' + utterance.audio.length + ' bytes');
|
|
308
|
+
} catch (err) {
|
|
309
|
+
console.log('[' + new Date().toISOString() + '] LISTEN Timeout: ' + err.message);
|
|
310
|
+
}
|
|
278
311
|
|
|
279
|
-
|
|
312
|
+
session.setCaptureEnabled(false);
|
|
313
|
+
}
|
|
280
314
|
|
|
281
315
|
if (!utterance) {
|
|
282
316
|
const promptUrl = await ttsService.generateSpeech("I didn't hear anything. Are you still there?", turnVoice);
|
|
@@ -355,8 +389,13 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
|
|
|
355
389
|
? await ttsService.generateSpeech(getRandomWaitingPhrase(), turnVoice)
|
|
356
390
|
: HOLD_MUSIC_URL;
|
|
357
391
|
if (!waiting) break;
|
|
358
|
-
// Caller can cut through the hold music / filler to add
|
|
359
|
-
|
|
392
|
+
// Caller can cut through the hold music / filler to add
|
|
393
|
+
// something - captured and queued as the next turn's input once
|
|
394
|
+
// the in-flight Claude query (already running, can't be
|
|
395
|
+
// cancelled mid-flight) finishes.
|
|
396
|
+
const barged = await playInterruptible(endpoint, session, clipUrl);
|
|
397
|
+
if (barged) {
|
|
398
|
+
pendingUtterance = barged;
|
|
360
399
|
waiting = false;
|
|
361
400
|
break;
|
|
362
401
|
}
|
|
@@ -377,13 +416,24 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
|
|
|
377
416
|
|
|
378
417
|
console.log('[' + new Date().toISOString() + '] CLAUDE Response received');
|
|
379
418
|
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
419
|
+
if (pendingUtterance) {
|
|
420
|
+
// Caller already interrupted during the wait and started saying
|
|
421
|
+
// something new - the answer to their original question is now
|
|
422
|
+
// moot. Skip playing it and let the next loop iteration process
|
|
423
|
+
// what they're actually saying now, instead of talking over/past it.
|
|
424
|
+
console.log('[' + new Date().toISOString() + '] VOICE: skipped (caller already interrupted with a new utterance)');
|
|
425
|
+
} else {
|
|
426
|
+
// Extract and play voice line with device voice
|
|
427
|
+
const voiceLine = extractVoiceLine(claudeResponse);
|
|
428
|
+
console.log('[' + new Date().toISOString() + '] VOICE: "' + voiceLine + '"');
|
|
429
|
+
|
|
430
|
+
const responseUrl = await ttsService.generateSpeech(voiceLine, turnVoice);
|
|
431
|
+
// Long answers are the usual thing people want to interrupt.
|
|
432
|
+
const responseBarge = await playInterruptible(endpoint, session, responseUrl);
|
|
433
|
+
if (responseBarge) {
|
|
434
|
+
pendingUtterance = responseBarge;
|
|
435
|
+
}
|
|
436
|
+
}
|
|
387
437
|
|
|
388
438
|
console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + ' complete');
|
|
389
439
|
}
|