@vellumai/assistant 0.12.0-dev.202609111819.3dcbf21 → 0.12.0-dev.202609111913.15f1f50

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/ARCHITECTURE.md CHANGED
@@ -609,7 +609,7 @@ Every phone call connects over Twilio Media Streams: the voice webhook emits `<C
609
609
 
610
610
  Transcription mode is selected once per session in `media-stream-stt-session.ts`:
611
611
 
612
- - **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials.
612
+ - **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials. While something is interruptible (an assistant turn in flight, thinking or speaking, or a completed turn's tail still playing from Twilio's buffer), caller speech arms the shared sustained-speech barge-in guard (`src/calls/barge-in-guard.ts`, the same accounting live voice uses: speech accumulates toward 250 ms, short gaps are tolerated, a run that is mostly silence resets) and every inbound frame feeds it; outside that window the guard is dropped, so the caller's own utterance never carries into a turn that starts before the local VAD ends it. Only a fired guard reaches `CallController.handleBargeIn`, which interrupts a turn in either phase and ignores an idle controller (the playing tail is cleared instead).
613
613
  - **Batch fallback**: otherwise the session segments turns with the energy-based `MediaTurnDetector` and transcribes each completed turn via the same role's batch API. Both halves of a call read the `telephony` role, which is why a role names its consumer rather than a boundary.
614
614
 
615
615
  Every phone turn runs the same two-leg triage as live voice through `startVoiceTurn` (`src/calls/voice-session-bridge.ts`): `call-controller.ts` opens on a toolless front-door leg (`routingLeg: "front-door"`, the `voiceFrontDoor` call site) and drives it through the shared `createFrontDoorLegCoordinator` (`src/calls/voice-leg-coordinator.ts`), which reads the stream through the verdict machine and sequences the hand-off (pause narration, abort the leg, resolve and speak the bridge, mark it as the floor holder, start the escalated leg pinned to the conversation's own model, re-arm narration); each driver supplies only a host for how text and the bridge are spoken, how a leg is started or aborted, and (live voice only) the speculative hold and commit. Phone has no partial transcripts, so the hold verdict is never taught and routing is escalate-only. The controller also passes the bridge's turn callbacks (tool activity is recorded as `tool_use_started` / `tool_use_completed` call events, persisted row ids ride the `assistant_spoke` event), `launchedAtMs` for dispatch timing, and a `voiceTelemetry` bag keyed by the call session with a `phone_inbound` / `phone_outbound` entry. Both drivers share the spoken progress narration cadence (`src/calls/voice-progress-cadence.ts`, tuned by `voice.frontModel.progress`): the cadence owns the tool-activity log, the triggers (an ops burst, a long operation completing, a full interval of audible silence with news, the `maxSilenceMs` heartbeat) and the generated or static phrase, while each driver supplies its own view of audible silence (live voice from its TTS queue and playback-tail estimate; the media-stream transport from `isPlaybackIdle()` and a running sum of sent frame durations) and how to speak a phrase.
package/openapi.yaml CHANGED
@@ -596,6 +596,16 @@ paths:
596
596
  type: string
597
597
  agent:
598
598
  type: string
599
+ requestedModel:
600
+ anyOf:
601
+ - type: string
602
+ - type: "null"
603
+ description: The model explicitly requested for this spawn, if any.
604
+ effectiveModel:
605
+ anyOf:
606
+ - type: string
607
+ - type: "null"
608
+ description: The top-level session model reported by the ACP adapter, if any.
599
609
  modelWarning:
600
610
  description: Why the requested model was not applied. The session is running on the agent's own model.
601
611
  type: string
@@ -603,6 +613,8 @@ paths:
603
613
  - acpSessionId
604
614
  - protocolSessionId
605
615
  - agent
616
+ - requestedModel
617
+ - effectiveModel
606
618
  additionalProperties: false
607
619
  /v1/activation/dismiss:
608
620
  post:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.0-dev.202609111819.3dcbf21",
3
+ "version": "0.12.0-dev.202609111913.15f1f50",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -26,7 +26,7 @@ afterAll(() => {
26
26
  });
27
27
 
28
28
  describe("always-loaded tool count", () => {
29
- test("should be exactly 11 with recall occupying the existing slot", async () => {
29
+ test("should be exactly 14 with recall occupying the existing slot", async () => {
30
30
  await initializeTools();
31
31
  const allDefs = getAllToolDefinitions();
32
32
 
@@ -48,10 +48,11 @@ describe("always-loaded tool count", () => {
48
48
  // connected — without a human in the loop, the guardian auto-approve
49
49
  // path would allow unchecked host command execution.
50
50
  //
51
- // `watch_retro_report` is here for the same reason the ui_surface tools are
52
- // NOT: a watch retrospective runs clientless, so it can only report through
53
- // a tool that survives this baseline. Its description and schema are kept
54
- // deliberately terse because that is the cost of the slot.
51
+ // `watch_retro_report` survives this baseline for the same reason the
52
+ // ui_* tools do: a watch retrospective runs clientless, so it can only
53
+ // report through a tool that survives this baseline. The ui_* tools are
54
+ // here because background UI surfaces persist and return instead of
55
+ // awaiting action, so they no longer need a connected client.
55
56
  const expectedNames = [
56
57
  "bash",
57
58
  "file_edit",
@@ -61,6 +62,9 @@ describe("always-loaded tool count", () => {
61
62
  "remember",
62
63
  "skill_execute",
63
64
  "skill_load",
65
+ "ui_dismiss",
66
+ "ui_show",
67
+ "ui_update",
64
68
  "watch_retro_report",
65
69
  "web_fetch",
66
70
  "web_search",
@@ -68,6 +72,6 @@ describe("always-loaded tool count", () => {
68
72
 
69
73
  expect(activeNames).toEqual(expectedNames);
70
74
  expect(activeNames.filter((name) => name === "recall")).toHaveLength(1);
71
- expect(activeTools.length).toBe(11);
75
+ expect(activeTools.length).toBe(14);
72
76
  });
73
77
  });
@@ -21,6 +21,7 @@ import {
21
21
  function makeContext(
22
22
  sent: AssistantEvent[] = [],
23
23
  channelCapabilities?: { channel: string; supportsDynamicUi: boolean },
24
+ opts?: { hasNoClient?: boolean },
24
25
  ): Conversation {
25
26
  return asConversation({
26
27
  conversationId: "session-1",
@@ -37,6 +38,7 @@ function makeContext(
37
38
  accumulatedSurfaceState: new Map<string, Record<string, unknown>>(),
38
39
  surfaceActionRequestIds: new Set<string>(),
39
40
  currentTurnSurfaces: [],
41
+ hasNoClient: opts?.hasNoClient ?? false,
40
42
  isProcessing: () => false,
41
43
  enqueueMessage: () => ({ queued: false, requestId: "req-1" }),
42
44
  getQueueDepth: () => 0,
@@ -65,6 +67,71 @@ describe("task_progress surface compatibility", () => {
65
67
  expect(sent).toHaveLength(0);
66
68
  });
67
69
 
70
+ test("persists a clientless choice from a non-rendering channel without waiting", async () => {
71
+ const sent: AssistantEvent[] = [];
72
+ const ctx = makeContext(
73
+ sent,
74
+ {
75
+ channel: "phone",
76
+ supportsDynamicUi: false,
77
+ },
78
+ { hasNoClient: true },
79
+ );
80
+
81
+ const result = await surfaceProxyResolver(ctx, "ui_show", {
82
+ surface_type: "choice",
83
+ title: "Choose a focus",
84
+ data: {
85
+ options: [
86
+ { id: "inbox", title: "Inbox" },
87
+ { id: "calendar", title: "Calendar" },
88
+ ],
89
+ },
90
+ });
91
+
92
+ expect(result.isError).toBe(false);
93
+ expect(result.yieldToUser).toBeUndefined();
94
+ const { surfaceId } = JSON.parse(result.content) as { surfaceId: string };
95
+ expect(ctx.currentTurnSurfaces.some((s) => s.surfaceId === surfaceId)).toBe(
96
+ true,
97
+ );
98
+ expect(ctx.pendingSurfaceActions.has(surfaceId)).toBe(false);
99
+ expect(sent.some((msg) => msg.type === "ui_surface_show")).toBe(true);
100
+ });
101
+
102
+ test("persists a clientless update in the current-turn surface snapshot", async () => {
103
+ const sent: AssistantEvent[] = [];
104
+ const ctx = makeContext(
105
+ sent,
106
+ {
107
+ channel: "phone",
108
+ supportsDynamicUi: false,
109
+ },
110
+ { hasNoClient: true },
111
+ );
112
+ const shown = await surfaceProxyResolver(ctx, "ui_show", {
113
+ surface_type: "card",
114
+ title: "Background work",
115
+ data: {
116
+ template: "task_progress",
117
+ templateData: { status: "in_progress", steps: [] },
118
+ },
119
+ });
120
+ const { surfaceId } = JSON.parse(shown.content) as { surfaceId: string };
121
+
122
+ const result = await surfaceProxyResolver(ctx, "ui_update", {
123
+ surface_id: surfaceId,
124
+ data: { templateData: { status: "completed" } },
125
+ });
126
+
127
+ expect(result.isError).toBe(false);
128
+ const data = ctx.currentTurnSurfaces.find((s) => s.surfaceId === surfaceId)
129
+ ?.data as CardSurfaceData;
130
+ expect((data.templateData as Record<string, unknown>).status).toBe(
131
+ "completed",
132
+ );
133
+ });
134
+
68
135
  test("blocks ui_update when channel lacks dynamic UI support", async () => {
69
136
  const sent: AssistantEvent[] = [];
70
137
  const ctx = makeContext(sent, {
@@ -249,14 +249,19 @@ describe("createResolveToolsCallback — toolContextPin", () => {
249
249
  });
250
250
  }
251
251
 
252
- test("control: without a pin, a clientless fork drops every client-gated tool from the wire", () => {
252
+ test("control: without a pin, a clientless fork drops every client-gated tool but ui_show", () => {
253
253
  projectedSkillToolNames = [];
254
254
  const resolve = createResolveToolsCallback(
255
255
  CLIENT_GATED_DEFS,
256
256
  clientlessExecutionCtx(),
257
257
  )!;
258
258
 
259
- expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual(["remember"]);
259
+ // ui_show stays on the wire: background UI surfaces persist and return
260
+ // instead of awaiting action, so they no longer need a connected client.
261
+ expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
262
+ "remember",
263
+ "ui_show",
264
+ ]);
260
265
  });
261
266
 
262
267
  test("a desktop-source pin restores the host/UI/client tool defs on the wire", () => {
@@ -291,7 +296,11 @@ describe("createResolveToolsCallback — toolContextPin", () => {
291
296
  }),
292
297
  )!;
293
298
 
294
- expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual(["remember"]);
299
+ // ui_show survives the clientless pin: it persists and returns.
300
+ expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
301
+ "remember",
302
+ "ui_show",
303
+ ]);
295
304
  });
296
305
 
297
306
  test("invariant: a pinned-in tool is on the wire but can never execute", async () => {
@@ -1,13 +1,10 @@
1
1
  /**
2
2
  * The gate that decides whether a watch retrospective can report at all.
3
3
  *
4
- * A retrospective runs as a `clientless` wake, which pins the turn
5
- * non-interactive. `conversation-tool-setup` gates the whole `ui_surface`
6
- * family on a client being present, so in that turn `ui_show` is not denied,
7
- * it is absent: a retrospective told to call it can only tell the user it
8
- * cannot. Nothing about the card's schema, its renderer, or the prompt's
9
- * wording reveals that, which is why it is asserted here against the real
10
- * registry rather than assumed anywhere else.
4
+ * A retrospective runs as a `clientless` wake, but its report continues to
5
+ * use the post-turn renderer so the report lands after its tool call has been
6
+ * persisted. The core UI tools remain available on that turn for other
7
+ * background work, so this file pins both contracts against the real registry.
11
8
  */
12
9
 
13
10
  import { afterAll, describe, expect, test } from "bun:test";
@@ -54,20 +51,14 @@ describe("watch retrospective tool availability", () => {
54
51
  ).toBe(true);
55
52
  });
56
53
 
57
- test("ui_show is absent from a clientless turn", async () => {
54
+ test("core UI tools stay available on a clientless turn", async () => {
58
55
  await initializeTools();
59
56
  const ctx = clientlessContext();
60
57
 
61
- // The reason the report does not go through `ui_show`. If this ever flips
62
- // to true, the extra tool above can be reconsidered; while it is false,
63
- // routing the retrospective's card through `ui_show` produces a turn that
64
- // cannot report at all.
65
- expect(isToolActiveForContext("ui_show", ctx)).toBe(false);
66
- expect(isToolActiveForContext("ui_update", ctx)).toBe(false);
67
- expect(isToolActiveForContext("ui_dismiss", ctx)).toBe(false);
58
+ for (const name of ["ui_show", "ui_update", "ui_dismiss"]) {
59
+ expect(isToolActiveForContext(name, ctx)).toBe(true);
60
+ }
68
61
 
69
- // And the gate really is about the client, not about the tools being
70
- // unregistered: with one attached, the same names are active.
71
62
  const withClient = clientfulContext();
72
63
  expect(isToolActiveForContext("ui_show", withClient)).toBe(true);
73
64
  });
@@ -246,22 +246,27 @@ describe("AcpSessionManager: model selection at spawn", () => {
246
246
  }): Promise<{
247
247
  state: AcpSessionState;
248
248
  sent: AssistantEvent[];
249
+ requestedModel?: string;
250
+ effectiveModel?: string;
249
251
  modelWarning?: string;
250
252
  }> {
251
253
  const manager = new AcpSessionManager(5);
252
254
  const sent: AssistantEvent[] = [];
253
- const { acpSessionId, modelWarning } = await manager.spawn(
254
- opts.agentId ?? "agent-model",
255
- { command: "echo", args: ["hi"], model: opts.agentModel },
256
- "task",
257
- "/tmp",
258
- opts.conversationId,
259
- (msg) => sent.push(msg),
260
- opts.requestedModel ? { model: opts.requestedModel } : {},
261
- );
255
+ const { acpSessionId, requestedModel, effectiveModel, modelWarning } =
256
+ await manager.spawn(
257
+ opts.agentId ?? "agent-model",
258
+ { command: "echo", args: ["hi"], model: opts.agentModel },
259
+ "task",
260
+ "/tmp",
261
+ opts.conversationId,
262
+ (msg) => sent.push(msg),
263
+ opts.requestedModel ? { model: opts.requestedModel } : {},
264
+ );
262
265
  return {
263
266
  state: manager.getStatus(acpSessionId) as AcpSessionState,
264
267
  sent,
268
+ requestedModel,
269
+ effectiveModel,
265
270
  modelWarning,
266
271
  };
267
272
  }
@@ -274,15 +279,36 @@ describe("AcpSessionManager: model selection at spawn", () => {
274
279
  test("state carries the model and options the adapter reported", async () => {
275
280
  scriptedConfigOptions = [[modelOption("opus")]];
276
281
 
277
- const { state } = await spawnWithModel({ conversationId: "conv-report" });
282
+ const { effectiveModel, state } = await spawnWithModel({
283
+ conversationId: "conv-report",
284
+ });
278
285
 
279
286
  expect(state.model).toBe("opus");
287
+ expect(effectiveModel).toBe("opus");
280
288
  expect(state.availableModels).toEqual(MODEL_OPTION_MODELS);
281
289
  // Nothing was requested and the adapter is already on a model, so it was
282
290
  // never asked to change.
283
291
  expect(setConfigOptionCalls).toEqual([]);
284
292
  });
285
293
 
294
+ test("the spawn result owns requested-model normalization", async () => {
295
+ scriptedConfigOptions = [[modelOption("default")]];
296
+
297
+ const manager = new AcpSessionManager(5);
298
+ const result = await manager.spawn(
299
+ "agent-model",
300
+ { command: "echo", args: ["hi"] },
301
+ "task",
302
+ "/tmp",
303
+ "conv-normalized-request",
304
+ () => {},
305
+ { model: " opus " },
306
+ );
307
+
308
+ expect(result.requestedModel).toBe("opus");
309
+ expect(selectedValue()).toBe("opus");
310
+ });
311
+
286
312
  test("the model event follows the spawned event", async () => {
287
313
  scriptedConfigOptions = [[modelOption("opus")]];
288
314
 
@@ -372,12 +398,13 @@ describe("AcpSessionManager: model selection at spawn", () => {
372
398
  // The adapter resolves the alias it was handed to a full model id.
373
399
  setConfigOptionResult = [modelOption("claude-opus-4-5")];
374
400
 
375
- const { state } = await spawnWithModel({
401
+ const { effectiveModel, state } = await spawnWithModel({
376
402
  conversationId: "conv-pin",
377
403
  requestedModel: "opus",
378
404
  });
379
405
 
380
406
  expect(state.model).toBe("claude-opus-4-5");
407
+ expect(effectiveModel).toBe("claude-opus-4-5");
381
408
  });
382
409
 
383
410
  test("an inherited model the adapter refuses warns nobody", async () => {
@@ -417,6 +444,7 @@ describe("AcpSessionManager: model selection at spawn", () => {
417
444
  expect(result.modelWarning).toBe(
418
445
  "Invalid value for config option model: nope",
419
446
  );
447
+ expect(result.effectiveModel).toBe("opus");
420
448
  const state = manager.getStatus(result.acpSessionId) as AcpSessionState;
421
449
  // The run is live on whatever the adapter chose for itself.
422
450
  expect(state.status).toBe("running");
@@ -284,9 +284,11 @@ export class AcpSessionManager {
284
284
  * The prompt is fired in the background — results stream via sessionUpdate
285
285
  * callbacks and completion/error messages are sent when the prompt finishes.
286
286
  *
287
+ * `requestedModel` is the normalized explicit request this spawn acted on.
288
+ * `effectiveModel` is the model the adapter reports for the live session.
287
289
  * `modelWarning` comes back when the adapter refused the model the session
288
- * was asked for: the run is live on the adapter's own default, and the
289
- * caller relays the reason rather than treating the spawn as failed.
290
+ * was asked for: the caller relays the reason rather than treating the spawn
291
+ * as failed.
290
292
  */
291
293
  async spawn(
292
294
  agentId: string,
@@ -300,6 +302,8 @@ export class AcpSessionManager {
300
302
  ): Promise<{
301
303
  acpSessionId: string;
302
304
  protocolSessionId: string;
305
+ requestedModel?: string;
306
+ effectiveModel?: string;
303
307
  modelWarning?: string;
304
308
  }> {
305
309
  this.assertCapacity();
@@ -429,6 +433,8 @@ export class AcpSessionManager {
429
433
  return {
430
434
  acpSessionId,
431
435
  protocolSessionId: state.acpSessionId,
436
+ requestedModel,
437
+ effectiveModel: state.model,
432
438
  ...(modelWarning ? { modelWarning } : {}),
433
439
  };
434
440
  }
@@ -0,0 +1,81 @@
1
+ import { describe, expect, test } from "bun:test";
2
+
3
+ import {
4
+ BARGE_IN_GAP_TOLERANCE_MS,
5
+ BARGE_IN_MAX_TOLERATED_SILENCE_RATIO,
6
+ createBargeInGuard,
7
+ } from "../barge-in-guard.js";
8
+
9
+ describe("createBargeInGuard", () => {
10
+ test("speech shorter than the threshold stays pending", () => {
11
+ const guard = createBargeInGuard(250);
12
+ expect(guard.track("speech", 100)).toBe("pending");
13
+ expect(guard.track("speech", 100)).toBe("pending");
14
+ expect(guard.speechMs).toBe(200);
15
+ });
16
+
17
+ test("sustained speech reaching the threshold fires once and stays fired", () => {
18
+ const guard = createBargeInGuard(250);
19
+ guard.track("speech", 200);
20
+ expect(guard.track("speech", 50)).toBe("fired");
21
+ expect(guard.track("silence", 1_000)).toBe("fired");
22
+ });
23
+
24
+ test("a brief sub-threshold gap does not reset the run", () => {
25
+ const guard = createBargeInGuard(250);
26
+ guard.track("speech", 150);
27
+ expect(guard.track("silence", 100)).toBe("pending");
28
+ expect(guard.speechMs).toBe(150);
29
+ expect(guard.track("speech", 100)).toBe("fired");
30
+ });
31
+
32
+ test("a gap of exactly the tolerance is still inside the run", () => {
33
+ const guard = createBargeInGuard(250);
34
+ guard.track("speech", 150);
35
+ expect(guard.track("silence", BARGE_IN_GAP_TOLERANCE_MS)).toBe("pending");
36
+ expect(guard.speechMs).toBe(150);
37
+ });
38
+
39
+ test("a gap longer than the tolerance resets the run", () => {
40
+ const guard = createBargeInGuard(250);
41
+ guard.track("speech", 150);
42
+ guard.track("silence", 150);
43
+ expect(guard.track("silence", 100)).toBe("reset");
44
+ expect(guard.speechMs).toBe(0);
45
+ expect(guard.track("speech", 200)).toBe("pending");
46
+ });
47
+
48
+ test("sparse periodic blips separated by boundary gaps never accumulate", () => {
49
+ const guard = createBargeInGuard(250);
50
+ // 10 ms blips every 200 ms: each gap is tolerated on its own, but the
51
+ // run's total silence outweighs the speech by the duty-cycle ceiling.
52
+ const cap = 250 * BARGE_IN_MAX_TOLERATED_SILENCE_RATIO;
53
+ let steps = 0;
54
+ let sawReset = false;
55
+ while (steps < 60) {
56
+ guard.track("speech", 10);
57
+ if (guard.track("silence", 200) === "reset") {
58
+ sawReset = true;
59
+ break;
60
+ }
61
+ steps += 1;
62
+ }
63
+ expect(sawReset).toBe(true);
64
+ expect(steps * 200).toBeLessThanOrEqual(cap + 200);
65
+ expect(guard.speechMs).toBe(0);
66
+ });
67
+
68
+ test("classified echo resets a partial run immediately", () => {
69
+ const guard = createBargeInGuard(250);
70
+ guard.track("speech", 200);
71
+ expect(guard.track("echo", 10)).toBe("reset");
72
+ expect(guard.speechMs).toBe(0);
73
+ });
74
+
75
+ test("a zero threshold fires on the first speech chunk", () => {
76
+ const guard = createBargeInGuard(0);
77
+ // With no speech required, any silence already outweighs it.
78
+ expect(guard.track("silence", 20)).toBe("reset");
79
+ expect(guard.track("speech", 20)).toBe("fired");
80
+ });
81
+ });
@@ -3101,7 +3101,7 @@ describe("call-controller", () => {
3101
3101
  controller.destroy();
3102
3102
  });
3103
3103
 
3104
- test("synthesized provider: stays 'processing' during synthesis latency, so barge-in cannot abort an inaudible turn", async () => {
3104
+ test("synthesized provider: stays 'processing' during synthesis latency until the play URL is sent", async () => {
3105
3105
  const cfg = loadConfig();
3106
3106
  cfg.services.tts.provider = "fish-audio";
3107
3107
  cfg.services.tts.providers["fish-audio"].referenceId = "fish-ref-123";
@@ -3142,11 +3142,11 @@ describe("call-controller", () => {
3142
3142
  await new Promise((r) => setTimeout(r, 20));
3143
3143
 
3144
3144
  // No audio has reached the caller yet (play URL not sent), so the controller
3145
- // must stay in `processing` — a barge-in here would abort a turn the caller
3146
- // cannot hear. See JARVIS-1232.
3145
+ // must stay in `processing`: the state tells the transport nothing is
3146
+ // audible, and only sustained caller speech (the transport's guard) may
3147
+ // cut the inaudible turn off. See JARVIS-1232.
3147
3148
  expect(relay.sentPlayUrls.length).toBe(0);
3148
3149
  expect(controller.getState()).toBe("processing");
3149
- expect(controller.handleBargeIn()).toBe(false);
3150
3150
 
3151
3151
  // Release audio → play URL sent → turn finishes and returns to idle.
3152
3152
  releaseChunk?.();
@@ -4665,10 +4665,11 @@ describe("call-controller", () => {
4665
4665
  controller.destroy();
4666
4666
  });
4667
4667
 
4668
- test("handleBargeIn returns false and does not abort while still processing (no output yet)", async () => {
4668
+ test("handleBargeIn accepts a barge-in while still processing and aborts the silent turn", async () => {
4669
4669
  // Simulate a turn stuck waiting for the processing lock: no tokens
4670
- // emitted, no completion. The controller must stay in `processing` and
4671
- // NOT flip to `speaking`, so barge-in can't abort a silent turn.
4670
+ // emitted, no completion. The controller stays in `processing`, and a
4671
+ // sustained caller barge-in (the transport's guard has already
4672
+ // vouched for it) cuts the thinking turn off before it speaks.
4672
4673
  mockStartVoiceTurn.mockImplementation(
4673
4674
  async (opts: {
4674
4675
  onTextDelta: (t: string) => void;
@@ -4693,18 +4694,17 @@ describe("call-controller", () => {
4693
4694
 
4694
4695
  const onAccepted = mock(() => {});
4695
4696
  const bargeResult = controller.handleBargeIn(onAccepted);
4696
- expect(bargeResult).toBe(false);
4697
- expect(onAccepted).not.toHaveBeenCalled();
4698
- // Still processing (not aborted), and no interrupt/end-of-turn token sent.
4699
- expect(controller.getState()).toBe("processing");
4697
+ expect(bargeResult).toBe(true);
4698
+ expect(onAccepted).toHaveBeenCalledTimes(1);
4699
+ // The turn is aborted; nothing was spoken, so no end-of-turn token.
4700
+ expect(controller.getState()).toBe("idle");
4700
4701
  const endTokens = relay.sentTokens.filter(
4701
4702
  (t) => t.last === true && t.token === "",
4702
4703
  );
4703
4704
  expect(endTokens.length).toBe(0);
4704
4705
 
4705
- // Cleanup: abort the pending turn
4706
- controller.destroy();
4707
4706
  await turnPromise.catch(() => {});
4707
+ controller.destroy();
4708
4708
  });
4709
4709
 
4710
4710
  test("stays in processing until first token, then flips to speaking and barge-in is accepted", async () => {
@@ -4746,9 +4746,7 @@ describe("call-controller", () => {
4746
4746
  await Promise.resolve();
4747
4747
  }
4748
4748
 
4749
- // Before any token: processing, barge-in ignored (turn not aborted).
4750
- expect(controller.getState()).toBe("processing");
4751
- expect(controller.handleBargeIn()).toBe(false);
4749
+ // Before any token: processing (no audio out yet).
4752
4750
  expect(controller.getState()).toBe("processing");
4753
4751
 
4754
4752
  // Release the first token → controller flips to speaking.
@@ -4766,7 +4764,7 @@ describe("call-controller", () => {
4766
4764
  await turnPromise.catch(() => {});
4767
4765
  });
4768
4766
 
4769
- test("buffering transport (media-stream): streamed tokens do not flip to speaking; barge-in is rejected and the turn still delivers", async () => {
4767
+ test("buffering transport (media-stream): streamed tokens do not flip to speaking; a barge-in still cuts the thinking turn off", async () => {
4770
4768
  // Simulates the media-stream transport: sendTextToken(.., false)
4771
4769
  // only buffers; audio starts later. The controller must stay in
4772
4770
  // `processing` until the transport's audio-start signal fires, so
@@ -4810,17 +4808,17 @@ describe("call-controller", () => {
4810
4808
  expect(controller.getState()).toBe("processing");
4811
4809
  expect(audioStartCallback).not.toBeNull();
4812
4810
 
4813
- // VAD speech-start mid-generation → barge-in rejected, turn intact.
4814
- expect(controller.handleBargeIn()).toBe(false);
4815
- expect(controller.getState()).toBe("processing");
4811
+ // Sustained caller speech mid-generation: the buffered, unspoken turn
4812
+ // is cut off like a spoken one would be, and its text is discarded.
4813
+ expect(controller.handleBargeIn()).toBe(true);
4814
+ expect(controller.getState()).toBe("idle");
4816
4815
 
4817
- // The turn completes and delivers its end-of-turn signal.
4818
- releaseTurn();
4819
- await turnPromise;
4816
+ await turnPromise.catch(() => {});
4817
+ // No end-of-turn token: nothing had been spoken.
4820
4818
  const endTokens = relay.sentTokens.filter(
4821
4819
  (t) => t.last === true && t.token === "",
4822
4820
  );
4823
- expect(endTokens.length).toBe(1);
4821
+ expect(endTokens.length).toBe(0);
4824
4822
 
4825
4823
  controller.destroy();
4826
4824
  });