claude-phone-local 2.0.12 → 2.0.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-phone-local",
3
- "version": "2.0.12",
3
+ "version": "2.0.13",
4
4
  "description": "Local/offline fork of NetworkChuck's claude-phone: talk to Claude Code over 3CX/SIP with faster-whisper STT + Piper TTS in one Docker container.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -278,6 +278,14 @@ class AudioForkSession extends EventEmitter {
278
278
  this.bargeInEnabled = false;
279
279
  console.log('[AUDIO-DEBUG] BARGE-IN detected (RMS=' + Math.round(stats.rms) + ') for ' + this.callUuid);
280
280
  this.emit('barge-in');
281
+ // Barge-in used to only stop playback - the words that triggered it,
282
+ // and everything the caller kept saying until the turn loop's next
283
+ // setCaptureEnabled(true), were silently discarded because capture
284
+ // stayed off the whole time. Turn capture on immediately so this
285
+ // chunk (already loud speech, just confirmed above) and what follows
286
+ // fall through into real utterance capture below instead of being
287
+ // dropped and forcing the caller to repeat themselves.
288
+ this.captureEnabled = true;
281
289
  }
282
290
  }
283
291
 
@@ -163,13 +163,20 @@ function extractVoiceLine(response) {
163
163
  *
164
164
  * FreeSWITCH plays to completion unless told otherwise, so to support barge-in
165
165
  * we arm the detector, then issue uuid_break the moment the caller starts
166
- * talking. Returns true if the caller cut in, so the loop can skip straight to
167
- * listening instead of finishing what it was saying.
166
+ * talking. AudioForkSession turns capture on itself right when barge-in
167
+ * fires (see audio-fork.js), so the words that triggered the interruption are
168
+ * already becoming a real utterance - previously they were discarded (capture
169
+ * stayed off during playback) and the caller had to repeat themselves after a
170
+ * fresh ready-beep on the next turn.
171
+ *
172
+ * Returns the captured utterance if the caller interrupted (so the turn loop
173
+ * can use it as their next input directly, no beep/re-prompt needed), or
174
+ * `null` if playback completed without interruption.
168
175
  */
169
176
  async function playInterruptible(endpoint, session, url) {
170
177
  if (!session) {
171
178
  await endpoint.play(url);
172
- return false;
179
+ return null;
173
180
  }
174
181
 
175
182
  let barged = false;
@@ -187,10 +194,25 @@ async function playInterruptible(endpoint, session, url) {
187
194
  session.setBargeInEnabled(false);
188
195
  session.removeListener('barge-in', onBarge);
189
196
  }
190
- if (barged) {
191
- console.log('[' + new Date().toISOString() + '] BARGE-IN: caller interrupted, listening now');
197
+
198
+ if (!barged) return null;
199
+
200
+ console.log('[' + new Date().toISOString() + '] BARGE-IN: caller interrupted, capturing what they said');
201
+ try {
202
+ // Capture already started the moment barge-in fired; this just waits for
203
+ // the utterance to finish (end-of-speech silence) rather than starting a
204
+ // fresh listen window that would miss the words already spoken.
205
+ const utterance = await session.waitForUtterance({ timeoutMs: 15000 });
206
+ // finalizeUtterance() resets utterance state but not captureEnabled -
207
+ // turn it off now, otherwise every chunk while we transcribe/query
208
+ // Claude keeps accumulating into a new stray utterance.
209
+ session.setCaptureEnabled(false);
210
+ return utterance;
211
+ } catch (err) {
212
+ console.log('[' + new Date().toISOString() + '] BARGE-IN: capture failed (' + err.message + '), falling back to a fresh listen');
213
+ session.setCaptureEnabled(false);
214
+ return null;
192
215
  }
193
- return barged;
194
216
  }
195
217
 
196
218
  /**
@@ -253,30 +275,42 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
253
275
  // Main conversation loop
254
276
  let turnCount = 0;
255
277
  const MAX_TURNS = 20;
278
+ // Speech captured by a barge-in during the previous turn's playback -
279
+ // consumed as this turn's input directly instead of a fresh ready-beep
280
+ // and listen window, which would otherwise make the caller repeat
281
+ // themselves after every interruption.
282
+ let pendingUtterance = null;
256
283
 
257
284
  while (turnCount < MAX_TURNS) {
258
285
  turnCount++;
259
286
  console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + '/' + MAX_TURNS);
260
287
 
261
- // READY BEEP
262
- try {
263
- await endpoint.play(READY_BEEP_URL);
264
- } catch (e) {
265
- console.log('[' + new Date().toISOString() + '] BEEP: Ready beep failed, continuing');
266
- }
288
+ let utterance = null;
289
+
290
+ if (pendingUtterance) {
291
+ console.log('[' + new Date().toISOString() + '] LISTEN Using speech captured during barge-in: ' + pendingUtterance.audio.length + ' bytes');
292
+ utterance = pendingUtterance;
293
+ pendingUtterance = null;
294
+ } else {
295
+ // READY BEEP
296
+ try {
297
+ await endpoint.play(READY_BEEP_URL);
298
+ } catch (e) {
299
+ console.log('[' + new Date().toISOString() + '] BEEP: Ready beep failed, continuing');
300
+ }
267
301
 
268
- session.setCaptureEnabled(true);
269
- console.log('[' + new Date().toISOString() + '] LISTEN Waiting for speech...');
302
+ session.setCaptureEnabled(true);
303
+ console.log('[' + new Date().toISOString() + '] LISTEN Waiting for speech...');
270
304
 
271
- let utterance = null;
272
- try {
273
- utterance = await session.waitForUtterance({ timeoutMs: 30000 });
274
- console.log('[' + new Date().toISOString() + '] LISTEN Got: ' + utterance.audio.length + ' bytes');
275
- } catch (err) {
276
- console.log('[' + new Date().toISOString() + '] LISTEN Timeout: ' + err.message);
277
- }
305
+ try {
306
+ utterance = await session.waitForUtterance({ timeoutMs: 30000 });
307
+ console.log('[' + new Date().toISOString() + '] LISTEN Got: ' + utterance.audio.length + ' bytes');
308
+ } catch (err) {
309
+ console.log('[' + new Date().toISOString() + '] LISTEN Timeout: ' + err.message);
310
+ }
278
311
 
279
- session.setCaptureEnabled(false);
312
+ session.setCaptureEnabled(false);
313
+ }
280
314
 
281
315
  if (!utterance) {
282
316
  const promptUrl = await ttsService.generateSpeech("I didn't hear anything. Are you still there?", turnVoice);
@@ -355,8 +389,13 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
355
389
  ? await ttsService.generateSpeech(getRandomWaitingPhrase(), turnVoice)
356
390
  : HOLD_MUSIC_URL;
357
391
  if (!waiting) break;
358
- // Caller can cut through the hold music / filler to add something.
359
- if (await playInterruptible(endpoint, session, clipUrl)) {
392
+ // Caller can cut through the hold music / filler to add
393
+ // something - captured and queued as the next turn's input once
394
+ // the in-flight Claude query (already running, can't be
395
+ // cancelled mid-flight) finishes.
396
+ const barged = await playInterruptible(endpoint, session, clipUrl);
397
+ if (barged) {
398
+ pendingUtterance = barged;
360
399
  waiting = false;
361
400
  break;
362
401
  }
@@ -377,13 +416,24 @@ async function conversationLoop(endpoint, dialog, callUuid, options, deviceConfi
377
416
 
378
417
  console.log('[' + new Date().toISOString() + '] CLAUDE Response received');
379
418
 
380
- // Extract and play voice line with device voice
381
- const voiceLine = extractVoiceLine(claudeResponse);
382
- console.log('[' + new Date().toISOString() + '] VOICE: "' + voiceLine + '"');
383
-
384
- const responseUrl = await ttsService.generateSpeech(voiceLine, turnVoice);
385
- // Long answers are the usual thing people want to interrupt.
386
- await playInterruptible(endpoint, session, responseUrl);
419
+ if (pendingUtterance) {
420
+ // Caller already interrupted during the wait and started saying
421
+ // something new - the answer to their original question is now
422
+ // moot. Skip playing it and let the next loop iteration process
423
+ // what they're actually saying now, instead of talking over/past it.
424
+ console.log('[' + new Date().toISOString() + '] VOICE: skipped (caller already interrupted with a new utterance)');
425
+ } else {
426
+ // Extract and play voice line with device voice
427
+ const voiceLine = extractVoiceLine(claudeResponse);
428
+ console.log('[' + new Date().toISOString() + '] VOICE: "' + voiceLine + '"');
429
+
430
+ const responseUrl = await ttsService.generateSpeech(voiceLine, turnVoice);
431
+ // Long answers are the usual thing people want to interrupt.
432
+ const responseBarge = await playInterruptible(endpoint, session, responseUrl);
433
+ if (responseBarge) {
434
+ pendingUtterance = responseBarge;
435
+ }
436
+ }
387
437
 
388
438
  console.log('[' + new Date().toISOString() + '] CONVERSATION Turn ' + turnCount + ' complete');
389
439
  }