@vellumai/assistant 0.12.0-dev.202609111819.3dcbf21 → 0.12.0-dev.202609111913.15f1f50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ARCHITECTURE.md +1 -1
- package/openapi.yaml +12 -0
- package/package.json +1 -1
- package/src/__tests__/always-loaded-tools-guard.test.ts +10 -6
- package/src/__tests__/conversation-surfaces-task-progress.test.ts +67 -0
- package/src/__tests__/subagent-tool-gate-mode.test.ts +12 -3
- package/src/__tests__/watch-retro-tool-availability.test.ts +8 -17
- package/src/acp/__tests__/session-manager.test.ts +39 -11
- package/src/acp/session-manager.ts +8 -2
- package/src/calls/__tests__/barge-in-guard.test.ts +81 -0
- package/src/calls/__tests__/call-controller.test.ts +22 -24
- package/src/calls/__tests__/media-stream-server-integration.test.ts +114 -24
- package/src/calls/barge-in-guard.ts +118 -0
- package/src/calls/call-controller.ts +29 -27
- package/src/calls/media-stream-server.ts +82 -30
- package/src/calls/media-stream-stt-session.ts +15 -0
- package/src/config/bundled-skills/acp/SKILL.md +2 -0
- package/src/config/bundled-skills/acp/TOOLS.json +1 -1
- package/src/daemon/__tests__/conversation-tool-setup.test.ts +32 -28
- package/src/daemon/conversation-surfaces.ts +7 -5
- package/src/daemon/conversation-tool-setup.ts +4 -8
- package/src/live-voice/live-voice-session.ts +26 -76
- package/src/runtime/routes/__tests__/acp-routes.test.ts +32 -2
- package/src/runtime/routes/acp-routes.ts +19 -1
- package/src/tools/acp/spawn.test.ts +61 -4
- package/src/tools/acp/spawn.ts +26 -15
- package/src/tools/watch/watch-retro-report.ts +0 -8
package/ARCHITECTURE.md
CHANGED
|
@@ -609,7 +609,7 @@ Every phone call connects over Twilio Media Streams: the voice webhook emits `<C
|
|
|
609
609
|
|
|
610
610
|
Transcription mode is selected once per session in `media-stream-stt-session.ts`:
|
|
611
611
|
|
|
612
|
-
- **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials.
|
|
612
|
+
- **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials. While something is interruptible (an assistant turn in flight, thinking or speaking, or a completed turn's tail still playing from Twilio's buffer), caller speech arms the shared sustained-speech barge-in guard (`src/calls/barge-in-guard.ts`, the same accounting live voice uses: speech accumulates toward 250 ms, short gaps are tolerated, a run that is mostly silence resets) and every inbound frame feeds it; outside that window the guard is dropped, so the caller's own utterance never carries into a turn that starts before the local VAD ends it. Only a fired guard reaches `CallController.handleBargeIn`, which interrupts a turn in either phase and ignores an idle controller (the playing tail is cleared instead).
|
|
613
613
|
- **Batch fallback**: otherwise the session segments turns with the energy-based `MediaTurnDetector` and transcribes each completed turn via the same role's batch API. Both halves of a call read the `telephony` role, which is why a role names its consumer rather than a boundary.
|
|
614
614
|
|
|
615
615
|
Every phone turn runs the same two-leg triage as live voice through `startVoiceTurn` (`src/calls/voice-session-bridge.ts`): `call-controller.ts` opens on a toolless front-door leg (`routingLeg: "front-door"`, the `voiceFrontDoor` call site) and drives it through the shared `createFrontDoorLegCoordinator` (`src/calls/voice-leg-coordinator.ts`), which reads the stream through the verdict machine and sequences the hand-off (pause narration, abort the leg, resolve and speak the bridge, mark it as the floor holder, start the escalated leg pinned to the conversation's own model, re-arm narration); each driver supplies only a host for how text and the bridge are spoken, how a leg is started or aborted, and (live voice only) the speculative hold and commit. Phone has no partial transcripts, so the hold verdict is never taught and routing is escalate-only. The controller also passes the bridge's turn callbacks (tool activity is recorded as `tool_use_started` / `tool_use_completed` call events, persisted row ids ride the `assistant_spoke` event), `launchedAtMs` for dispatch timing, and a `voiceTelemetry` bag keyed by the call session with a `phone_inbound` / `phone_outbound` entry. Both drivers share the spoken progress narration cadence (`src/calls/voice-progress-cadence.ts`, tuned by `voice.frontModel.progress`): the cadence owns the tool-activity log, the triggers (an ops burst, a long operation completing, a full interval of audible silence with news, the `maxSilenceMs` heartbeat) and the generated or static phrase, while each driver supplies its own view of audible silence (live voice from its TTS queue and playback-tail estimate; the media-stream transport from `isPlaybackIdle()` and a running sum of sent frame durations) and how to speak a phrase.
|
package/openapi.yaml
CHANGED
|
@@ -596,6 +596,16 @@ paths:
|
|
|
596
596
|
type: string
|
|
597
597
|
agent:
|
|
598
598
|
type: string
|
|
599
|
+
requestedModel:
|
|
600
|
+
anyOf:
|
|
601
|
+
- type: string
|
|
602
|
+
- type: "null"
|
|
603
|
+
description: The model explicitly requested for this spawn, if any.
|
|
604
|
+
effectiveModel:
|
|
605
|
+
anyOf:
|
|
606
|
+
- type: string
|
|
607
|
+
- type: "null"
|
|
608
|
+
description: The top-level session model reported by the ACP adapter, if any.
|
|
599
609
|
modelWarning:
|
|
600
610
|
description: Why the requested model was not applied. The session is running on the agent's own model.
|
|
601
611
|
type: string
|
|
@@ -603,6 +613,8 @@ paths:
|
|
|
603
613
|
- acpSessionId
|
|
604
614
|
- protocolSessionId
|
|
605
615
|
- agent
|
|
616
|
+
- requestedModel
|
|
617
|
+
- effectiveModel
|
|
606
618
|
additionalProperties: false
|
|
607
619
|
/v1/activation/dismiss:
|
|
608
620
|
post:
|
package/package.json
CHANGED
|
@@ -26,7 +26,7 @@ afterAll(() => {
|
|
|
26
26
|
});
|
|
27
27
|
|
|
28
28
|
describe("always-loaded tool count", () => {
|
|
29
|
-
test("should be exactly
|
|
29
|
+
test("should be exactly 14 with recall occupying the existing slot", async () => {
|
|
30
30
|
await initializeTools();
|
|
31
31
|
const allDefs = getAllToolDefinitions();
|
|
32
32
|
|
|
@@ -48,10 +48,11 @@ describe("always-loaded tool count", () => {
|
|
|
48
48
|
// connected — without a human in the loop, the guardian auto-approve
|
|
49
49
|
// path would allow unchecked host command execution.
|
|
50
50
|
//
|
|
51
|
-
// `watch_retro_report`
|
|
52
|
-
//
|
|
53
|
-
// a tool that survives this baseline.
|
|
54
|
-
//
|
|
51
|
+
// `watch_retro_report` survives this baseline for the same reason the
|
|
52
|
+
// ui_* tools do: a watch retrospective runs clientless, so it can only
|
|
53
|
+
// report through a tool that survives this baseline. The ui_* tools are
|
|
54
|
+
// here because background UI surfaces persist and return instead of
|
|
55
|
+
// awaiting action, so they no longer need a connected client.
|
|
55
56
|
const expectedNames = [
|
|
56
57
|
"bash",
|
|
57
58
|
"file_edit",
|
|
@@ -61,6 +62,9 @@ describe("always-loaded tool count", () => {
|
|
|
61
62
|
"remember",
|
|
62
63
|
"skill_execute",
|
|
63
64
|
"skill_load",
|
|
65
|
+
"ui_dismiss",
|
|
66
|
+
"ui_show",
|
|
67
|
+
"ui_update",
|
|
64
68
|
"watch_retro_report",
|
|
65
69
|
"web_fetch",
|
|
66
70
|
"web_search",
|
|
@@ -68,6 +72,6 @@ describe("always-loaded tool count", () => {
|
|
|
68
72
|
|
|
69
73
|
expect(activeNames).toEqual(expectedNames);
|
|
70
74
|
expect(activeNames.filter((name) => name === "recall")).toHaveLength(1);
|
|
71
|
-
expect(activeTools.length).toBe(
|
|
75
|
+
expect(activeTools.length).toBe(14);
|
|
72
76
|
});
|
|
73
77
|
});
|
|
@@ -21,6 +21,7 @@ import {
|
|
|
21
21
|
function makeContext(
|
|
22
22
|
sent: AssistantEvent[] = [],
|
|
23
23
|
channelCapabilities?: { channel: string; supportsDynamicUi: boolean },
|
|
24
|
+
opts?: { hasNoClient?: boolean },
|
|
24
25
|
): Conversation {
|
|
25
26
|
return asConversation({
|
|
26
27
|
conversationId: "session-1",
|
|
@@ -37,6 +38,7 @@ function makeContext(
|
|
|
37
38
|
accumulatedSurfaceState: new Map<string, Record<string, unknown>>(),
|
|
38
39
|
surfaceActionRequestIds: new Set<string>(),
|
|
39
40
|
currentTurnSurfaces: [],
|
|
41
|
+
hasNoClient: opts?.hasNoClient ?? false,
|
|
40
42
|
isProcessing: () => false,
|
|
41
43
|
enqueueMessage: () => ({ queued: false, requestId: "req-1" }),
|
|
42
44
|
getQueueDepth: () => 0,
|
|
@@ -65,6 +67,71 @@ describe("task_progress surface compatibility", () => {
|
|
|
65
67
|
expect(sent).toHaveLength(0);
|
|
66
68
|
});
|
|
67
69
|
|
|
70
|
+
test("persists a clientless choice from a non-rendering channel without waiting", async () => {
|
|
71
|
+
const sent: AssistantEvent[] = [];
|
|
72
|
+
const ctx = makeContext(
|
|
73
|
+
sent,
|
|
74
|
+
{
|
|
75
|
+
channel: "phone",
|
|
76
|
+
supportsDynamicUi: false,
|
|
77
|
+
},
|
|
78
|
+
{ hasNoClient: true },
|
|
79
|
+
);
|
|
80
|
+
|
|
81
|
+
const result = await surfaceProxyResolver(ctx, "ui_show", {
|
|
82
|
+
surface_type: "choice",
|
|
83
|
+
title: "Choose a focus",
|
|
84
|
+
data: {
|
|
85
|
+
options: [
|
|
86
|
+
{ id: "inbox", title: "Inbox" },
|
|
87
|
+
{ id: "calendar", title: "Calendar" },
|
|
88
|
+
],
|
|
89
|
+
},
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
expect(result.isError).toBe(false);
|
|
93
|
+
expect(result.yieldToUser).toBeUndefined();
|
|
94
|
+
const { surfaceId } = JSON.parse(result.content) as { surfaceId: string };
|
|
95
|
+
expect(ctx.currentTurnSurfaces.some((s) => s.surfaceId === surfaceId)).toBe(
|
|
96
|
+
true,
|
|
97
|
+
);
|
|
98
|
+
expect(ctx.pendingSurfaceActions.has(surfaceId)).toBe(false);
|
|
99
|
+
expect(sent.some((msg) => msg.type === "ui_surface_show")).toBe(true);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
test("persists a clientless update in the current-turn surface snapshot", async () => {
|
|
103
|
+
const sent: AssistantEvent[] = [];
|
|
104
|
+
const ctx = makeContext(
|
|
105
|
+
sent,
|
|
106
|
+
{
|
|
107
|
+
channel: "phone",
|
|
108
|
+
supportsDynamicUi: false,
|
|
109
|
+
},
|
|
110
|
+
{ hasNoClient: true },
|
|
111
|
+
);
|
|
112
|
+
const shown = await surfaceProxyResolver(ctx, "ui_show", {
|
|
113
|
+
surface_type: "card",
|
|
114
|
+
title: "Background work",
|
|
115
|
+
data: {
|
|
116
|
+
template: "task_progress",
|
|
117
|
+
templateData: { status: "in_progress", steps: [] },
|
|
118
|
+
},
|
|
119
|
+
});
|
|
120
|
+
const { surfaceId } = JSON.parse(shown.content) as { surfaceId: string };
|
|
121
|
+
|
|
122
|
+
const result = await surfaceProxyResolver(ctx, "ui_update", {
|
|
123
|
+
surface_id: surfaceId,
|
|
124
|
+
data: { templateData: { status: "completed" } },
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
expect(result.isError).toBe(false);
|
|
128
|
+
const data = ctx.currentTurnSurfaces.find((s) => s.surfaceId === surfaceId)
|
|
129
|
+
?.data as CardSurfaceData;
|
|
130
|
+
expect((data.templateData as Record<string, unknown>).status).toBe(
|
|
131
|
+
"completed",
|
|
132
|
+
);
|
|
133
|
+
});
|
|
134
|
+
|
|
68
135
|
test("blocks ui_update when channel lacks dynamic UI support", async () => {
|
|
69
136
|
const sent: AssistantEvent[] = [];
|
|
70
137
|
const ctx = makeContext(sent, {
|
|
@@ -249,14 +249,19 @@ describe("createResolveToolsCallback — toolContextPin", () => {
|
|
|
249
249
|
});
|
|
250
250
|
}
|
|
251
251
|
|
|
252
|
-
test("control: without a pin, a clientless fork drops every client-gated tool
|
|
252
|
+
test("control: without a pin, a clientless fork drops every client-gated tool but ui_show", () => {
|
|
253
253
|
projectedSkillToolNames = [];
|
|
254
254
|
const resolve = createResolveToolsCallback(
|
|
255
255
|
CLIENT_GATED_DEFS,
|
|
256
256
|
clientlessExecutionCtx(),
|
|
257
257
|
)!;
|
|
258
258
|
|
|
259
|
-
|
|
259
|
+
// ui_show stays on the wire: background UI surfaces persist and return
|
|
260
|
+
// instead of awaiting action, so they no longer need a connected client.
|
|
261
|
+
expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
|
|
262
|
+
"remember",
|
|
263
|
+
"ui_show",
|
|
264
|
+
]);
|
|
260
265
|
});
|
|
261
266
|
|
|
262
267
|
test("a desktop-source pin restores the host/UI/client tool defs on the wire", () => {
|
|
@@ -291,7 +296,11 @@ describe("createResolveToolsCallback — toolContextPin", () => {
|
|
|
291
296
|
}),
|
|
292
297
|
)!;
|
|
293
298
|
|
|
294
|
-
|
|
299
|
+
// ui_show survives the clientless pin: it persists and returns.
|
|
300
|
+
expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
|
|
301
|
+
"remember",
|
|
302
|
+
"ui_show",
|
|
303
|
+
]);
|
|
295
304
|
});
|
|
296
305
|
|
|
297
306
|
test("invariant: a pinned-in tool is on the wire but can never execute", async () => {
|
|
@@ -1,13 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The gate that decides whether a watch retrospective can report at all.
|
|
3
3
|
*
|
|
4
|
-
* A retrospective runs as a `clientless` wake,
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* cannot. Nothing about the card's schema, its renderer, or the prompt's
|
|
9
|
-
* wording reveals that, which is why it is asserted here against the real
|
|
10
|
-
* registry rather than assumed anywhere else.
|
|
4
|
+
* A retrospective runs as a `clientless` wake, but its report continues to
|
|
5
|
+
* use the post-turn renderer so the report lands after its tool call has been
|
|
6
|
+
* persisted. The core UI tools remain available on that turn for other
|
|
7
|
+
* background work, so this file pins both contracts against the real registry.
|
|
11
8
|
*/
|
|
12
9
|
|
|
13
10
|
import { afterAll, describe, expect, test } from "bun:test";
|
|
@@ -54,20 +51,14 @@ describe("watch retrospective tool availability", () => {
|
|
|
54
51
|
).toBe(true);
|
|
55
52
|
});
|
|
56
53
|
|
|
57
|
-
test("
|
|
54
|
+
test("core UI tools stay available on a clientless turn", async () => {
|
|
58
55
|
await initializeTools();
|
|
59
56
|
const ctx = clientlessContext();
|
|
60
57
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
// cannot report at all.
|
|
65
|
-
expect(isToolActiveForContext("ui_show", ctx)).toBe(false);
|
|
66
|
-
expect(isToolActiveForContext("ui_update", ctx)).toBe(false);
|
|
67
|
-
expect(isToolActiveForContext("ui_dismiss", ctx)).toBe(false);
|
|
58
|
+
for (const name of ["ui_show", "ui_update", "ui_dismiss"]) {
|
|
59
|
+
expect(isToolActiveForContext(name, ctx)).toBe(true);
|
|
60
|
+
}
|
|
68
61
|
|
|
69
|
-
// And the gate really is about the client, not about the tools being
|
|
70
|
-
// unregistered: with one attached, the same names are active.
|
|
71
62
|
const withClient = clientfulContext();
|
|
72
63
|
expect(isToolActiveForContext("ui_show", withClient)).toBe(true);
|
|
73
64
|
});
|
|
@@ -246,22 +246,27 @@ describe("AcpSessionManager: model selection at spawn", () => {
|
|
|
246
246
|
}): Promise<{
|
|
247
247
|
state: AcpSessionState;
|
|
248
248
|
sent: AssistantEvent[];
|
|
249
|
+
requestedModel?: string;
|
|
250
|
+
effectiveModel?: string;
|
|
249
251
|
modelWarning?: string;
|
|
250
252
|
}> {
|
|
251
253
|
const manager = new AcpSessionManager(5);
|
|
252
254
|
const sent: AssistantEvent[] = [];
|
|
253
|
-
const { acpSessionId, modelWarning } =
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
255
|
+
const { acpSessionId, requestedModel, effectiveModel, modelWarning } =
|
|
256
|
+
await manager.spawn(
|
|
257
|
+
opts.agentId ?? "agent-model",
|
|
258
|
+
{ command: "echo", args: ["hi"], model: opts.agentModel },
|
|
259
|
+
"task",
|
|
260
|
+
"/tmp",
|
|
261
|
+
opts.conversationId,
|
|
262
|
+
(msg) => sent.push(msg),
|
|
263
|
+
opts.requestedModel ? { model: opts.requestedModel } : {},
|
|
264
|
+
);
|
|
262
265
|
return {
|
|
263
266
|
state: manager.getStatus(acpSessionId) as AcpSessionState,
|
|
264
267
|
sent,
|
|
268
|
+
requestedModel,
|
|
269
|
+
effectiveModel,
|
|
265
270
|
modelWarning,
|
|
266
271
|
};
|
|
267
272
|
}
|
|
@@ -274,15 +279,36 @@ describe("AcpSessionManager: model selection at spawn", () => {
|
|
|
274
279
|
test("state carries the model and options the adapter reported", async () => {
|
|
275
280
|
scriptedConfigOptions = [[modelOption("opus")]];
|
|
276
281
|
|
|
277
|
-
const { state } = await spawnWithModel({
|
|
282
|
+
const { effectiveModel, state } = await spawnWithModel({
|
|
283
|
+
conversationId: "conv-report",
|
|
284
|
+
});
|
|
278
285
|
|
|
279
286
|
expect(state.model).toBe("opus");
|
|
287
|
+
expect(effectiveModel).toBe("opus");
|
|
280
288
|
expect(state.availableModels).toEqual(MODEL_OPTION_MODELS);
|
|
281
289
|
// Nothing was requested and the adapter is already on a model, so it was
|
|
282
290
|
// never asked to change.
|
|
283
291
|
expect(setConfigOptionCalls).toEqual([]);
|
|
284
292
|
});
|
|
285
293
|
|
|
294
|
+
test("the spawn result owns requested-model normalization", async () => {
|
|
295
|
+
scriptedConfigOptions = [[modelOption("default")]];
|
|
296
|
+
|
|
297
|
+
const manager = new AcpSessionManager(5);
|
|
298
|
+
const result = await manager.spawn(
|
|
299
|
+
"agent-model",
|
|
300
|
+
{ command: "echo", args: ["hi"] },
|
|
301
|
+
"task",
|
|
302
|
+
"/tmp",
|
|
303
|
+
"conv-normalized-request",
|
|
304
|
+
() => {},
|
|
305
|
+
{ model: " opus " },
|
|
306
|
+
);
|
|
307
|
+
|
|
308
|
+
expect(result.requestedModel).toBe("opus");
|
|
309
|
+
expect(selectedValue()).toBe("opus");
|
|
310
|
+
});
|
|
311
|
+
|
|
286
312
|
test("the model event follows the spawned event", async () => {
|
|
287
313
|
scriptedConfigOptions = [[modelOption("opus")]];
|
|
288
314
|
|
|
@@ -372,12 +398,13 @@ describe("AcpSessionManager: model selection at spawn", () => {
|
|
|
372
398
|
// The adapter resolves the alias it was handed to a full model id.
|
|
373
399
|
setConfigOptionResult = [modelOption("claude-opus-4-5")];
|
|
374
400
|
|
|
375
|
-
const { state } = await spawnWithModel({
|
|
401
|
+
const { effectiveModel, state } = await spawnWithModel({
|
|
376
402
|
conversationId: "conv-pin",
|
|
377
403
|
requestedModel: "opus",
|
|
378
404
|
});
|
|
379
405
|
|
|
380
406
|
expect(state.model).toBe("claude-opus-4-5");
|
|
407
|
+
expect(effectiveModel).toBe("claude-opus-4-5");
|
|
381
408
|
});
|
|
382
409
|
|
|
383
410
|
test("an inherited model the adapter refuses warns nobody", async () => {
|
|
@@ -417,6 +444,7 @@ describe("AcpSessionManager: model selection at spawn", () => {
|
|
|
417
444
|
expect(result.modelWarning).toBe(
|
|
418
445
|
"Invalid value for config option model: nope",
|
|
419
446
|
);
|
|
447
|
+
expect(result.effectiveModel).toBe("opus");
|
|
420
448
|
const state = manager.getStatus(result.acpSessionId) as AcpSessionState;
|
|
421
449
|
// The run is live on whatever the adapter chose for itself.
|
|
422
450
|
expect(state.status).toBe("running");
|
|
@@ -284,9 +284,11 @@ export class AcpSessionManager {
|
|
|
284
284
|
* The prompt is fired in the background — results stream via sessionUpdate
|
|
285
285
|
* callbacks and completion/error messages are sent when the prompt finishes.
|
|
286
286
|
*
|
|
287
|
+
* `requestedModel` is the normalized explicit request this spawn acted on.
|
|
288
|
+
* `effectiveModel` is the model the adapter reports for the live session.
|
|
287
289
|
* `modelWarning` comes back when the adapter refused the model the session
|
|
288
|
-
* was asked for: the
|
|
289
|
-
*
|
|
290
|
+
* was asked for: the caller relays the reason rather than treating the spawn
|
|
291
|
+
* as failed.
|
|
290
292
|
*/
|
|
291
293
|
async spawn(
|
|
292
294
|
agentId: string,
|
|
@@ -300,6 +302,8 @@ export class AcpSessionManager {
|
|
|
300
302
|
): Promise<{
|
|
301
303
|
acpSessionId: string;
|
|
302
304
|
protocolSessionId: string;
|
|
305
|
+
requestedModel?: string;
|
|
306
|
+
effectiveModel?: string;
|
|
303
307
|
modelWarning?: string;
|
|
304
308
|
}> {
|
|
305
309
|
this.assertCapacity();
|
|
@@ -429,6 +433,8 @@ export class AcpSessionManager {
|
|
|
429
433
|
return {
|
|
430
434
|
acpSessionId,
|
|
431
435
|
protocolSessionId: state.acpSessionId,
|
|
436
|
+
requestedModel,
|
|
437
|
+
effectiveModel: state.model,
|
|
432
438
|
...(modelWarning ? { modelWarning } : {}),
|
|
433
439
|
};
|
|
434
440
|
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { describe, expect, test } from "bun:test";
|
|
2
|
+
|
|
3
|
+
import {
|
|
4
|
+
BARGE_IN_GAP_TOLERANCE_MS,
|
|
5
|
+
BARGE_IN_MAX_TOLERATED_SILENCE_RATIO,
|
|
6
|
+
createBargeInGuard,
|
|
7
|
+
} from "../barge-in-guard.js";
|
|
8
|
+
|
|
9
|
+
describe("createBargeInGuard", () => {
|
|
10
|
+
test("speech shorter than the threshold stays pending", () => {
|
|
11
|
+
const guard = createBargeInGuard(250);
|
|
12
|
+
expect(guard.track("speech", 100)).toBe("pending");
|
|
13
|
+
expect(guard.track("speech", 100)).toBe("pending");
|
|
14
|
+
expect(guard.speechMs).toBe(200);
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
test("sustained speech reaching the threshold fires once and stays fired", () => {
|
|
18
|
+
const guard = createBargeInGuard(250);
|
|
19
|
+
guard.track("speech", 200);
|
|
20
|
+
expect(guard.track("speech", 50)).toBe("fired");
|
|
21
|
+
expect(guard.track("silence", 1_000)).toBe("fired");
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
test("a brief sub-threshold gap does not reset the run", () => {
|
|
25
|
+
const guard = createBargeInGuard(250);
|
|
26
|
+
guard.track("speech", 150);
|
|
27
|
+
expect(guard.track("silence", 100)).toBe("pending");
|
|
28
|
+
expect(guard.speechMs).toBe(150);
|
|
29
|
+
expect(guard.track("speech", 100)).toBe("fired");
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
test("a gap of exactly the tolerance is still inside the run", () => {
|
|
33
|
+
const guard = createBargeInGuard(250);
|
|
34
|
+
guard.track("speech", 150);
|
|
35
|
+
expect(guard.track("silence", BARGE_IN_GAP_TOLERANCE_MS)).toBe("pending");
|
|
36
|
+
expect(guard.speechMs).toBe(150);
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
test("a gap longer than the tolerance resets the run", () => {
|
|
40
|
+
const guard = createBargeInGuard(250);
|
|
41
|
+
guard.track("speech", 150);
|
|
42
|
+
guard.track("silence", 150);
|
|
43
|
+
expect(guard.track("silence", 100)).toBe("reset");
|
|
44
|
+
expect(guard.speechMs).toBe(0);
|
|
45
|
+
expect(guard.track("speech", 200)).toBe("pending");
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test("sparse periodic blips separated by boundary gaps never accumulate", () => {
|
|
49
|
+
const guard = createBargeInGuard(250);
|
|
50
|
+
// 10 ms blips every 200 ms: each gap is tolerated on its own, but the
|
|
51
|
+
// run's total silence outweighs the speech by the duty-cycle ceiling.
|
|
52
|
+
const cap = 250 * BARGE_IN_MAX_TOLERATED_SILENCE_RATIO;
|
|
53
|
+
let steps = 0;
|
|
54
|
+
let sawReset = false;
|
|
55
|
+
while (steps < 60) {
|
|
56
|
+
guard.track("speech", 10);
|
|
57
|
+
if (guard.track("silence", 200) === "reset") {
|
|
58
|
+
sawReset = true;
|
|
59
|
+
break;
|
|
60
|
+
}
|
|
61
|
+
steps += 1;
|
|
62
|
+
}
|
|
63
|
+
expect(sawReset).toBe(true);
|
|
64
|
+
expect(steps * 200).toBeLessThanOrEqual(cap + 200);
|
|
65
|
+
expect(guard.speechMs).toBe(0);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
test("classified echo resets a partial run immediately", () => {
|
|
69
|
+
const guard = createBargeInGuard(250);
|
|
70
|
+
guard.track("speech", 200);
|
|
71
|
+
expect(guard.track("echo", 10)).toBe("reset");
|
|
72
|
+
expect(guard.speechMs).toBe(0);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
test("a zero threshold fires on the first speech chunk", () => {
|
|
76
|
+
const guard = createBargeInGuard(0);
|
|
77
|
+
// With no speech required, any silence already outweighs it.
|
|
78
|
+
expect(guard.track("silence", 20)).toBe("reset");
|
|
79
|
+
expect(guard.track("speech", 20)).toBe("fired");
|
|
80
|
+
});
|
|
81
|
+
});
|
|
@@ -3101,7 +3101,7 @@ describe("call-controller", () => {
|
|
|
3101
3101
|
controller.destroy();
|
|
3102
3102
|
});
|
|
3103
3103
|
|
|
3104
|
-
test("synthesized provider: stays 'processing' during synthesis latency
|
|
3104
|
+
test("synthesized provider: stays 'processing' during synthesis latency until the play URL is sent", async () => {
|
|
3105
3105
|
const cfg = loadConfig();
|
|
3106
3106
|
cfg.services.tts.provider = "fish-audio";
|
|
3107
3107
|
cfg.services.tts.providers["fish-audio"].referenceId = "fish-ref-123";
|
|
@@ -3142,11 +3142,11 @@ describe("call-controller", () => {
|
|
|
3142
3142
|
await new Promise((r) => setTimeout(r, 20));
|
|
3143
3143
|
|
|
3144
3144
|
// No audio has reached the caller yet (play URL not sent), so the controller
|
|
3145
|
-
// must stay in `processing
|
|
3146
|
-
//
|
|
3145
|
+
// must stay in `processing`: the state tells the transport nothing is
|
|
3146
|
+
// audible, and only sustained caller speech (the transport's guard) may
|
|
3147
|
+
// cut the inaudible turn off. See JARVIS-1232.
|
|
3147
3148
|
expect(relay.sentPlayUrls.length).toBe(0);
|
|
3148
3149
|
expect(controller.getState()).toBe("processing");
|
|
3149
|
-
expect(controller.handleBargeIn()).toBe(false);
|
|
3150
3150
|
|
|
3151
3151
|
// Release audio → play URL sent → turn finishes and returns to idle.
|
|
3152
3152
|
releaseChunk?.();
|
|
@@ -4665,10 +4665,11 @@ describe("call-controller", () => {
|
|
|
4665
4665
|
controller.destroy();
|
|
4666
4666
|
});
|
|
4667
4667
|
|
|
4668
|
-
test("handleBargeIn
|
|
4668
|
+
test("handleBargeIn accepts a barge-in while still processing and aborts the silent turn", async () => {
|
|
4669
4669
|
// Simulate a turn stuck waiting for the processing lock: no tokens
|
|
4670
|
-
// emitted, no completion. The controller
|
|
4671
|
-
//
|
|
4670
|
+
// emitted, no completion. The controller stays in `processing`, and a
|
|
4671
|
+
// sustained caller barge-in (the transport's guard has already
|
|
4672
|
+
// vouched for it) cuts the thinking turn off before it speaks.
|
|
4672
4673
|
mockStartVoiceTurn.mockImplementation(
|
|
4673
4674
|
async (opts: {
|
|
4674
4675
|
onTextDelta: (t: string) => void;
|
|
@@ -4693,18 +4694,17 @@ describe("call-controller", () => {
|
|
|
4693
4694
|
|
|
4694
4695
|
const onAccepted = mock(() => {});
|
|
4695
4696
|
const bargeResult = controller.handleBargeIn(onAccepted);
|
|
4696
|
-
expect(bargeResult).toBe(
|
|
4697
|
-
expect(onAccepted).
|
|
4698
|
-
//
|
|
4699
|
-
expect(controller.getState()).toBe("
|
|
4697
|
+
expect(bargeResult).toBe(true);
|
|
4698
|
+
expect(onAccepted).toHaveBeenCalledTimes(1);
|
|
4699
|
+
// The turn is aborted; nothing was spoken, so no end-of-turn token.
|
|
4700
|
+
expect(controller.getState()).toBe("idle");
|
|
4700
4701
|
const endTokens = relay.sentTokens.filter(
|
|
4701
4702
|
(t) => t.last === true && t.token === "",
|
|
4702
4703
|
);
|
|
4703
4704
|
expect(endTokens.length).toBe(0);
|
|
4704
4705
|
|
|
4705
|
-
// Cleanup: abort the pending turn
|
|
4706
|
-
controller.destroy();
|
|
4707
4706
|
await turnPromise.catch(() => {});
|
|
4707
|
+
controller.destroy();
|
|
4708
4708
|
});
|
|
4709
4709
|
|
|
4710
4710
|
test("stays in processing until first token, then flips to speaking and barge-in is accepted", async () => {
|
|
@@ -4746,9 +4746,7 @@ describe("call-controller", () => {
|
|
|
4746
4746
|
await Promise.resolve();
|
|
4747
4747
|
}
|
|
4748
4748
|
|
|
4749
|
-
// Before any token: processing
|
|
4750
|
-
expect(controller.getState()).toBe("processing");
|
|
4751
|
-
expect(controller.handleBargeIn()).toBe(false);
|
|
4749
|
+
// Before any token: processing (no audio out yet).
|
|
4752
4750
|
expect(controller.getState()).toBe("processing");
|
|
4753
4751
|
|
|
4754
4752
|
// Release the first token → controller flips to speaking.
|
|
@@ -4766,7 +4764,7 @@ describe("call-controller", () => {
|
|
|
4766
4764
|
await turnPromise.catch(() => {});
|
|
4767
4765
|
});
|
|
4768
4766
|
|
|
4769
|
-
test("buffering transport (media-stream): streamed tokens do not flip to speaking; barge-in
|
|
4767
|
+
test("buffering transport (media-stream): streamed tokens do not flip to speaking; a barge-in still cuts the thinking turn off", async () => {
|
|
4770
4768
|
// Simulates the media-stream transport: sendTextToken(.., false)
|
|
4771
4769
|
// only buffers; audio starts later. The controller must stay in
|
|
4772
4770
|
// `processing` until the transport's audio-start signal fires, so
|
|
@@ -4810,17 +4808,17 @@ describe("call-controller", () => {
|
|
|
4810
4808
|
expect(controller.getState()).toBe("processing");
|
|
4811
4809
|
expect(audioStartCallback).not.toBeNull();
|
|
4812
4810
|
|
|
4813
|
-
//
|
|
4814
|
-
|
|
4815
|
-
expect(controller.
|
|
4811
|
+
// Sustained caller speech mid-generation: the buffered, unspoken turn
|
|
4812
|
+
// is cut off like a spoken one would be, and its text is discarded.
|
|
4813
|
+
expect(controller.handleBargeIn()).toBe(true);
|
|
4814
|
+
expect(controller.getState()).toBe("idle");
|
|
4816
4815
|
|
|
4817
|
-
|
|
4818
|
-
|
|
4819
|
-
await turnPromise;
|
|
4816
|
+
await turnPromise.catch(() => {});
|
|
4817
|
+
// No end-of-turn token: nothing had been spoken.
|
|
4820
4818
|
const endTokens = relay.sentTokens.filter(
|
|
4821
4819
|
(t) => t.last === true && t.token === "",
|
|
4822
4820
|
);
|
|
4823
|
-
expect(endTokens.length).toBe(
|
|
4821
|
+
expect(endTokens.length).toBe(0);
|
|
4824
4822
|
|
|
4825
4823
|
controller.destroy();
|
|
4826
4824
|
});
|