@mastra/livekit 0.3.0 → 0.3.1-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +6 -4
- package/README.md +1 -1
- package/dist/bridge.d.ts.map +1 -1
- package/dist/index.cjs +172 -134
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +168 -121
- package/dist/index.js.map +1 -1
- package/dist/llm-plugin.d.ts.map +1 -1
- package/dist/plugin-entry.cjs +199 -172
- package/dist/plugin-entry.cjs.map +1 -1
- package/dist/plugin-entry.js +195 -165
- package/dist/plugin-entry.js.map +1 -1
- package/dist/remote-C9K3UzKv.js +629 -0
- package/dist/remote-C9K3UzKv.js.map +1 -0
- package/dist/remote-D0Y5P6e4.cjs +658 -0
- package/dist/remote-D0Y5P6e4.cjs.map +1 -0
- package/dist/worker-entry.cjs +621 -538
- package/dist/worker-entry.cjs.map +1 -1
- package/dist/worker-entry.js +618 -529
- package/dist/worker-entry.js.map +1 -1
- package/dist/workflow-generator-B67QrY8L.cjs +209 -0
- package/dist/workflow-generator-B67QrY8L.cjs.map +1 -0
- package/dist/workflow-generator-BtfClQcM.js +180 -0
- package/dist/workflow-generator-BtfClQcM.js.map +1 -0
- package/package.json +14 -14
- package/CHANGELOG.md +0 -426
- package/dist/chunk-2E3MTAOA.js +0 -133
- package/dist/chunk-2E3MTAOA.js.map +0 -1
- package/dist/chunk-4O7IN74Y.js +0 -568
- package/dist/chunk-4O7IN74Y.js.map +0 -1
- package/dist/chunk-DBVKNDAQ.cjs +0 -574
- package/dist/chunk-DBVKNDAQ.cjs.map +0 -1
- package/dist/chunk-MWTEZOBS.cjs +0 -139
- package/dist/chunk-MWTEZOBS.cjs.map +0 -1
package/dist/worker-entry.js
CHANGED
|
@@ -1,586 +1,675 @@
|
|
|
1
|
-
import { parseSessionMetadata, createWorkflowReplyGenerator
|
|
2
|
-
import { createMastraVoiceAgent } from
|
|
3
|
-
|
|
4
|
-
import {
|
|
5
|
-
import { RequestContext } from
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
8
|
-
|
|
9
|
-
|
|
1
|
+
import { r as parseSessionMetadata, t as createWorkflowReplyGenerator } from "./workflow-generator-BtfClQcM.js";
|
|
2
|
+
import { i as chatContextToMessages, r as createMastraVoiceAgent, t as createRemoteAgentReplyGenerator } from "./remote-C9K3UzKv.js";
|
|
3
|
+
import { randomUUID } from "crypto";
|
|
4
|
+
import { InferenceRunner, ServerOptions, cli, defineAgent, metrics, voice } from "@livekit/agents";
|
|
5
|
+
import { RequestContext } from "@mastra/core/request-context";
|
|
6
|
+
import { SpanType, getOrCreateSpan } from "@mastra/core/observability";
|
|
7
|
+
import { fileURLToPath } from "url";
|
|
8
|
+
//#region src/observability.ts
|
|
10
9
|
function modelMeta(metadata) {
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
10
|
+
const out = {};
|
|
11
|
+
if (metadata?.modelProvider) out.modelProvider = metadata.modelProvider;
|
|
12
|
+
if (metadata?.modelName) out.modelName = metadata.modelName;
|
|
13
|
+
return out;
|
|
15
14
|
}
|
|
16
|
-
|
|
17
|
-
|
|
15
|
+
const ms = (n) => `${Math.round(n)}ms`;
|
|
16
|
+
const secs = (n) => `${(n / 1e3).toFixed(1)}s`;
|
|
17
|
+
/**
|
|
18
|
+
* Maps a LiveKit pipeline metric to an event-span name and the latency/usage fields worth
|
|
19
|
+
* recording. The name carries the headline number so it reads in the trace timeline without
|
|
20
|
+
* opening the span. Metrics this release does not surface (interruption, EOT inference, avatar)
|
|
21
|
+
* return `undefined` — they still feed the session usage roll-up via the ModelUsageCollector.
|
|
22
|
+
*/
|
|
18
23
|
function describeMetric(metric) {
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
totalTokens: metric.totalTokens,
|
|
85
|
-
...modelMeta(metric.metadata)
|
|
86
|
-
}
|
|
87
|
-
};
|
|
88
|
-
default:
|
|
89
|
-
return void 0;
|
|
90
|
-
}
|
|
24
|
+
switch (metric.type) {
|
|
25
|
+
case "eou_metrics": return {
|
|
26
|
+
name: `eou ${ms(metric.endOfUtteranceDelayMs)}`,
|
|
27
|
+
data: {
|
|
28
|
+
endOfUtteranceDelayMs: metric.endOfUtteranceDelayMs,
|
|
29
|
+
transcriptionDelayMs: metric.transcriptionDelayMs,
|
|
30
|
+
onUserTurnCompletedDelayMs: metric.onUserTurnCompletedDelayMs
|
|
31
|
+
}
|
|
32
|
+
};
|
|
33
|
+
case "stt_metrics": return {
|
|
34
|
+
name: `stt ${secs(metric.audioDurationMs)}`,
|
|
35
|
+
data: {
|
|
36
|
+
audioDurationMs: metric.audioDurationMs,
|
|
37
|
+
durationMs: metric.durationMs,
|
|
38
|
+
streamed: metric.streamed,
|
|
39
|
+
...modelMeta(metric.metadata)
|
|
40
|
+
}
|
|
41
|
+
};
|
|
42
|
+
case "llm_metrics": return {
|
|
43
|
+
name: `llm ttft ${ms(metric.ttftMs)}`,
|
|
44
|
+
data: {
|
|
45
|
+
ttftMs: metric.ttftMs,
|
|
46
|
+
durationMs: metric.durationMs,
|
|
47
|
+
tokensPerSecond: metric.tokensPerSecond,
|
|
48
|
+
promptTokens: metric.promptTokens,
|
|
49
|
+
completionTokens: metric.completionTokens,
|
|
50
|
+
totalTokens: metric.totalTokens,
|
|
51
|
+
cancelled: metric.cancelled,
|
|
52
|
+
...modelMeta(metric.metadata)
|
|
53
|
+
}
|
|
54
|
+
};
|
|
55
|
+
case "tts_metrics": return {
|
|
56
|
+
name: `tts ttfb ${ms(metric.ttfbMs)}`,
|
|
57
|
+
data: {
|
|
58
|
+
ttfbMs: metric.ttfbMs,
|
|
59
|
+
durationMs: metric.durationMs,
|
|
60
|
+
audioDurationMs: metric.audioDurationMs,
|
|
61
|
+
charactersCount: metric.charactersCount,
|
|
62
|
+
cancelled: metric.cancelled,
|
|
63
|
+
streamed: metric.streamed,
|
|
64
|
+
...modelMeta(metric.metadata)
|
|
65
|
+
}
|
|
66
|
+
};
|
|
67
|
+
case "vad_metrics": return {
|
|
68
|
+
name: "vad",
|
|
69
|
+
data: {
|
|
70
|
+
idleTimeMs: metric.idleTimeMs,
|
|
71
|
+
inferenceCount: metric.inferenceCount,
|
|
72
|
+
inferenceDurationTotalMs: metric.inferenceDurationTotalMs
|
|
73
|
+
}
|
|
74
|
+
};
|
|
75
|
+
case "realtime_model_metrics": return {
|
|
76
|
+
name: `realtime ttft ${ms(metric.ttftMs)}`,
|
|
77
|
+
data: {
|
|
78
|
+
ttftMs: metric.ttftMs,
|
|
79
|
+
durationMs: metric.durationMs,
|
|
80
|
+
tokensPerSecond: metric.tokensPerSecond,
|
|
81
|
+
promptTokens: metric.inputTokens,
|
|
82
|
+
completionTokens: metric.outputTokens,
|
|
83
|
+
totalTokens: metric.totalTokens,
|
|
84
|
+
...modelMeta(metric.metadata)
|
|
85
|
+
}
|
|
86
|
+
};
|
|
87
|
+
default: return;
|
|
88
|
+
}
|
|
91
89
|
}
|
|
90
|
+
/**
|
|
91
|
+
* Opens a `voice call` trace for one LiveKit session and bridges LiveKit's voice-pipeline
|
|
92
|
+
* metrics into Mastra observability.
|
|
93
|
+
*
|
|
94
|
+
* The returned root span groups everything about the call: each conversation turn's Mastra
|
|
95
|
+
* agent run (nested via {@link VoiceCallObservability.tracingContext}) plus an event span for
|
|
96
|
+
* every STT, TTS, end-of-utterance, VAD, and LLM-latency metric LiveKit emits. Metrics are
|
|
97
|
+
* point-in-time, so they're recorded as event spans (no duration) with the value in the name.
|
|
98
|
+
* The root closes on the session `close` event with a per-model usage roll-up (token, character,
|
|
99
|
+
* and audio totals for the whole call) from a `ModelUsageCollector`.
|
|
100
|
+
*
|
|
101
|
+
* Returns `undefined` when the Mastra instance has no observability configured, so callers
|
|
102
|
+
* can treat instrumentation as a no-op without branching on config.
|
|
103
|
+
*/
|
|
92
104
|
function startVoiceCallObservability(options) {
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
105
|
+
const span = getOrCreateSpan({
|
|
106
|
+
mastra: options.mastra,
|
|
107
|
+
type: SpanType.GENERIC,
|
|
108
|
+
name: "voice call",
|
|
109
|
+
requestContext: options.requestContext,
|
|
110
|
+
metadata: {
|
|
111
|
+
agentId: options.agentId,
|
|
112
|
+
roomName: options.roomName,
|
|
113
|
+
...options.metadata.threadId ? { threadId: options.metadata.threadId } : {},
|
|
114
|
+
...options.metadata.resourceId ? { resourceId: options.metadata.resourceId } : {}
|
|
115
|
+
}
|
|
116
|
+
});
|
|
117
|
+
if (!span) return void 0;
|
|
118
|
+
const usage = new metrics.ModelUsageCollector();
|
|
119
|
+
let finalized = false;
|
|
120
|
+
const finalize = (opts) => {
|
|
121
|
+
if (finalized) return;
|
|
122
|
+
finalized = true;
|
|
123
|
+
const summary = usage.flatten();
|
|
124
|
+
if (opts?.error) span.error({
|
|
125
|
+
error: opts.error instanceof Error ? opts.error : new Error(String(opts.error)),
|
|
126
|
+
metadata: { usage: summary }
|
|
127
|
+
});
|
|
128
|
+
else span.end({ output: { usage: summary } });
|
|
129
|
+
};
|
|
130
|
+
return {
|
|
131
|
+
span,
|
|
132
|
+
tracingContext: { currentSpan: span },
|
|
133
|
+
attach(session) {
|
|
134
|
+
session.on(voice.AgentSessionEventTypes.MetricsCollected, (event) => {
|
|
135
|
+
usage.collect(event.metrics);
|
|
136
|
+
const described = describeMetric(event.metrics);
|
|
137
|
+
if (!described) return;
|
|
138
|
+
span.createEventSpan({
|
|
139
|
+
type: SpanType.GENERIC,
|
|
140
|
+
name: described.name,
|
|
141
|
+
output: described.data
|
|
142
|
+
});
|
|
143
|
+
});
|
|
144
|
+
session.on(voice.AgentSessionEventTypes.Close, (event) => {
|
|
145
|
+
finalize({ error: event?.error ?? void 0 });
|
|
146
|
+
});
|
|
147
|
+
},
|
|
148
|
+
finalize
|
|
149
|
+
};
|
|
137
150
|
}
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
151
|
+
//#endregion
|
|
152
|
+
//#region src/voice-thread.ts
|
|
153
|
+
/**
|
|
154
|
+
* Creates the call's memory thread up front when it doesn't exist yet, titled and tagged
|
|
155
|
+
* so it reads as a voice call in Studio's thread list instead of an untitled thread that
|
|
156
|
+
* only appears after the first exchange.
|
|
157
|
+
*/
|
|
158
|
+
async function ensureVoiceCallThread({ memory, threadId, resourceId, roomName }) {
|
|
159
|
+
if (await memory.getThreadById({ threadId })) return;
|
|
160
|
+
await memory.createThread({
|
|
161
|
+
threadId,
|
|
162
|
+
resourceId,
|
|
163
|
+
title: "Voice call",
|
|
164
|
+
metadata: {
|
|
165
|
+
source: "livekit",
|
|
166
|
+
roomName
|
|
167
|
+
}
|
|
168
|
+
});
|
|
152
169
|
}
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
}) {
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
170
|
+
/**
|
|
171
|
+
* Persists the spoken greeting as an assistant message. The greeting is spoken via TTS
|
|
172
|
+
* without going through the Mastra agent, so without this the saved thread would start
|
|
173
|
+
* at the caller's first words instead of being a faithful call transcript.
|
|
174
|
+
*/
|
|
175
|
+
async function persistSpokenGreeting({ memory, threadId, resourceId, greeting }) {
|
|
176
|
+
if (!greeting.trim()) return;
|
|
177
|
+
const message = {
|
|
178
|
+
id: randomUUID(),
|
|
179
|
+
role: "assistant",
|
|
180
|
+
type: "text",
|
|
181
|
+
createdAt: /* @__PURE__ */ new Date(),
|
|
182
|
+
threadId,
|
|
183
|
+
resourceId,
|
|
184
|
+
content: {
|
|
185
|
+
format: 2,
|
|
186
|
+
parts: [{
|
|
187
|
+
type: "text",
|
|
188
|
+
text: greeting
|
|
189
|
+
}],
|
|
190
|
+
metadata: {
|
|
191
|
+
source: "voice",
|
|
192
|
+
kind: "greeting"
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
};
|
|
196
|
+
await memory.saveMessages({ messages: [message] });
|
|
174
197
|
}
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
198
|
+
//#endregion
|
|
199
|
+
//#region src/worker-setup.ts
|
|
200
|
+
/**
|
|
201
|
+
* Coordination between worker definition and worker startup. LiveKit plugins register
|
|
202
|
+
* their inference runners at module-import time, and the AgentServer only spawns the
|
|
203
|
+
* inference process for runners registered before it starts — so plugin imports kicked
|
|
204
|
+
* off by `createLiveKitWorker()` must complete before `runLiveKitWorker()` boots the
|
|
205
|
+
* server.
|
|
206
|
+
*/
|
|
207
|
+
const pendingSetup = [];
|
|
208
|
+
const requestedEouMethods = /* @__PURE__ */ new Set();
|
|
179
209
|
function queueWorkerSetup(setup) {
|
|
180
|
-
|
|
210
|
+
pendingSetup.push(setup);
|
|
181
211
|
}
|
|
182
212
|
function workerSetupComplete() {
|
|
183
|
-
|
|
213
|
+
return Promise.allSettled(pendingSetup);
|
|
184
214
|
}
|
|
185
215
|
function requestEouMethod(method) {
|
|
186
|
-
|
|
216
|
+
requestedEouMethods.add(method);
|
|
187
217
|
}
|
|
188
218
|
function isEouMethodRequested(method) {
|
|
189
|
-
|
|
219
|
+
return requestedEouMethods.has(method);
|
|
190
220
|
}
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
221
|
+
//#endregion
|
|
222
|
+
//#region src/worker.ts
|
|
223
|
+
const EOU_METHODS = {
|
|
224
|
+
english: "lk_end_of_utterance_en",
|
|
225
|
+
multilingual: "lk_end_of_utterance_multilingual"
|
|
196
226
|
};
|
|
197
227
|
async function loadSileroVad() {
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
);
|
|
206
|
-
}
|
|
207
|
-
return silero.VAD.load();
|
|
228
|
+
let silero;
|
|
229
|
+
try {
|
|
230
|
+
silero = await import("@livekit/agents-plugin-silero");
|
|
231
|
+
} catch (error) {
|
|
232
|
+
throw new Error("@mastra/livekit: voice activity detection requires '@livekit/agents-plugin-silero'. Install it, pass your own `vad` instance, or set `vad: false`.", { cause: error });
|
|
233
|
+
}
|
|
234
|
+
return silero.VAD.load();
|
|
208
235
|
}
|
|
209
236
|
async function loadTurnDetector(kind) {
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
);
|
|
218
|
-
}
|
|
219
|
-
return kind === "english" ? new plugin.turnDetector.EnglishModel() : new plugin.turnDetector.MultilingualModel();
|
|
237
|
+
let plugin;
|
|
238
|
+
try {
|
|
239
|
+
plugin = await import("@livekit/agents-plugin-livekit");
|
|
240
|
+
} catch (error) {
|
|
241
|
+
throw new Error(`@mastra/livekit: turnDetection '${kind}' requires '@livekit/agents-plugin-livekit'. Install it or use a built-in mode like 'vad' or 'stt'.`, { cause: error });
|
|
242
|
+
}
|
|
243
|
+
return kind === "english" ? new plugin.turnDetector.EnglishModel() : new plugin.turnDetector.MultilingualModel();
|
|
220
244
|
}
|
|
221
245
|
async function resolveMastraAgent(options, args) {
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
return options.mastra.getAgentById(ref);
|
|
232
|
-
} catch {
|
|
233
|
-
return options.mastra.getAgent(ref);
|
|
234
|
-
}
|
|
246
|
+
let ref = typeof options.agent === "function" ? await options.agent(args) : options.agent;
|
|
247
|
+
ref ??= args.metadata.agentId;
|
|
248
|
+
if (!ref) throw new Error("@mastra/livekit: no Mastra agent specified. Set `agent` on createLiveKitWorker or pass `agentId` in the dispatch metadata (e.g. via liveKitConnectionRoute).");
|
|
249
|
+
if (typeof ref !== "string") return ref;
|
|
250
|
+
try {
|
|
251
|
+
return options.mastra.getAgentById(ref);
|
|
252
|
+
} catch {
|
|
253
|
+
return options.mastra.getAgent(ref);
|
|
254
|
+
}
|
|
235
255
|
}
|
|
236
256
|
async function resolveWorkflow(options, args) {
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
if (typeof ref !== "string") return { workflow: ref };
|
|
243
|
-
return { workflow: options.mastra.getWorkflowById(ref) };
|
|
257
|
+
const resolver = options.workflow;
|
|
258
|
+
const ref = typeof resolver === "function" ? await resolver(args) : resolver ?? "";
|
|
259
|
+
if (!ref) throw new Error("@mastra/livekit: no workflow specified. Set `workflow` on createLiveKitWorker.");
|
|
260
|
+
if (typeof ref !== "string") return { workflow: ref };
|
|
261
|
+
return { workflow: options.mastra.getWorkflowById(ref) };
|
|
244
262
|
}
|
|
245
263
|
function resolveMemory(options, mastraAgent, args, roomName) {
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
264
|
+
if (options.memory === false) return false;
|
|
265
|
+
if (typeof options.memory === "function") return options.memory({
|
|
266
|
+
...args,
|
|
267
|
+
roomName
|
|
268
|
+
});
|
|
269
|
+
if (mastraAgent && !mastraAgent.hasOwnMemory()) return false;
|
|
270
|
+
const thread = args.metadata.threadId ?? roomName;
|
|
271
|
+
return {
|
|
272
|
+
thread,
|
|
273
|
+
resource: args.metadata.resourceId ?? thread
|
|
274
|
+
};
|
|
251
275
|
}
|
|
252
276
|
async function resolveMemoryInstance(options, mastraAgent, args, requestContext) {
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
277
|
+
if (mastraAgent) return await mastraAgent.getMemory({ requestContext }) ?? null;
|
|
278
|
+
const resolver = options.memoryInstance;
|
|
279
|
+
if (!resolver) return null;
|
|
280
|
+
const instance = typeof resolver === "function" ? await resolver(args) : resolver;
|
|
281
|
+
if (!instance) return null;
|
|
282
|
+
instance.__registerMastra(options.mastra);
|
|
283
|
+
if (!instance.hasOwnStorage) {
|
|
284
|
+
const storage = options.mastra.getStorage();
|
|
285
|
+
if (storage) instance.setStorage(storage);
|
|
286
|
+
}
|
|
287
|
+
return instance;
|
|
264
288
|
}
|
|
265
289
|
function buildTurnHandling(options, turnDetection) {
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
// `turnHandling.preemptiveGeneration` if the latency win matters more than exact
|
|
272
|
-
// thread history.
|
|
273
|
-
preemptiveGeneration: { enabled: false },
|
|
274
|
-
...options.turnHandling
|
|
275
|
-
};
|
|
290
|
+
return {
|
|
291
|
+
...turnDetection ? { turnDetection } : {},
|
|
292
|
+
preemptiveGeneration: { enabled: false },
|
|
293
|
+
...options.turnHandling
|
|
294
|
+
};
|
|
276
295
|
}
|
|
277
296
|
async function resolveInstructions(mastraAgent, requestContext) {
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
297
|
+
try {
|
|
298
|
+
const instructions = await mastraAgent.getInstructions({ requestContext });
|
|
299
|
+
return typeof instructions === "string" ? instructions : void 0;
|
|
300
|
+
} catch {
|
|
301
|
+
return;
|
|
302
|
+
}
|
|
284
303
|
}
|
|
304
|
+
/**
|
|
305
|
+
* Resolves the effective greeting from the canonical `configuration.greeting` and the deprecated
|
|
306
|
+
* top-level `greeting` / `persistGreeting` options. The legacy options are the base so existing
|
|
307
|
+
* worker configs keep working unchanged; `configuration.greeting` overrides field-by-field. Exported
|
|
308
|
+
* for unit testing.
|
|
309
|
+
*/
|
|
285
310
|
function resolveGreetingConfig(options) {
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
311
|
+
return {
|
|
312
|
+
text: options.greeting,
|
|
313
|
+
persist: options.persistGreeting,
|
|
314
|
+
...options.configuration?.greeting
|
|
315
|
+
};
|
|
291
316
|
}
|
|
317
|
+
/**
|
|
318
|
+
* Resolves the greeting text for a call: returns a fixed string as-is, or invokes the resolver form
|
|
319
|
+
* with the call context to produce a per-tenant greeting. Empty / whitespace-nothing results
|
|
320
|
+
* normalize to `undefined` (no greeting). Exported for unit testing.
|
|
321
|
+
*/
|
|
292
322
|
async function resolveGreetingText(text, context) {
|
|
293
|
-
|
|
294
|
-
return resolved || void 0;
|
|
323
|
+
return (typeof text === "function" ? await text(context) : text) || void 0;
|
|
295
324
|
}
|
|
325
|
+
/**
|
|
326
|
+
* Resolves a per-call session component (STT / TTS): invokes the `configuration` resolver with the
|
|
327
|
+
* call context, falling back to the static top-level option when there is no resolver or it
|
|
328
|
+
* resolves to `undefined`. Exported for unit testing.
|
|
329
|
+
*/
|
|
296
330
|
async function resolveSessionComponent(resolver, fallback, context) {
|
|
297
|
-
|
|
298
|
-
return resolved ?? fallback;
|
|
331
|
+
return (resolver ? await resolver(context) : void 0) ?? fallback;
|
|
299
332
|
}
|
|
333
|
+
/**
|
|
334
|
+
* Speaks the session's opening greeting, honoring the interruption / playout options. Takes the
|
|
335
|
+
* already-resolved greeting (see {@link resolveGreetingText}) and returns the LiveKit `SpeechHandle`,
|
|
336
|
+
* or `undefined` when there is no greeting text. Extracted and exported so the greeting behavior is
|
|
337
|
+
* unit-testable without a live room.
|
|
338
|
+
*/
|
|
300
339
|
async function speakGreeting(session, greeting) {
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
);
|
|
306
|
-
if (greeting.awaitPlayout) {
|
|
307
|
-
await handle.waitForPlayout().catch(() => {
|
|
308
|
-
});
|
|
309
|
-
}
|
|
310
|
-
return handle;
|
|
340
|
+
if (!greeting.text) return void 0;
|
|
341
|
+
const handle = session.say(greeting.text, greeting.allowInterruptions === void 0 ? void 0 : { allowInterruptions: greeting.allowInterruptions });
|
|
342
|
+
if (greeting.awaitPlayout) await handle.waitForPlayout().catch(() => {});
|
|
343
|
+
return handle;
|
|
311
344
|
}
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
345
|
+
/** Default tool name the worker watches for to end the call. Matches `createEndCallTool`'s default. */
|
|
346
|
+
const DEFAULT_END_CALL_TOOL = "endCall";
|
|
347
|
+
/** Default shutdown reason recorded when the agent ends the call. */
|
|
348
|
+
const DEFAULT_END_CALL_REASON = "agent ended call";
|
|
349
|
+
/** Default safety cap on waiting for the agent's closing words to finish before hanging up. */
|
|
350
|
+
const DEFAULT_END_CALL_MAX_WAIT_MS = 3e4;
|
|
351
|
+
/**
|
|
352
|
+
* Default post-playout drain before the room is deleted. LiveKit's "playout completed" is
|
|
353
|
+
* worker-local; the caller's client still holds network + jitter-buffer audio, so hanging up the
|
|
354
|
+
* instant the state clears clips the tail of the goodbye.
|
|
355
|
+
*/
|
|
356
|
+
const DEFAULT_END_CALL_DRAIN_MS = 800;
|
|
357
|
+
/** Agent states where the agent is still busy producing / playing a reply (not done speaking). */
|
|
358
|
+
const AGENT_BUSY_STATES = /* @__PURE__ */ new Set(["thinking", "speaking"]);
|
|
359
|
+
/**
|
|
360
|
+
* Resolves once the agent is no longer producing or playing a reply — i.e. its state has left
|
|
361
|
+
* `thinking`/`speaking` for `listening`/`idle`. Returns immediately when it's already idle. A
|
|
362
|
+
* `maxWaitMs` safety cap guarantees it resolves even if the speaking state never clears. Used before
|
|
363
|
+
* an agent-initiated hang-up so the closing words play out fully instead of being cut off by the
|
|
364
|
+
* session close. Exported for unit testing.
|
|
365
|
+
*/
|
|
317
366
|
function waitForAgentDoneSpeaking(session, maxWaitMs = DEFAULT_END_CALL_MAX_WAIT_MS) {
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
367
|
+
if (!AGENT_BUSY_STATES.has(session.agentState)) return Promise.resolve();
|
|
368
|
+
return new Promise((resolve) => {
|
|
369
|
+
let timer;
|
|
370
|
+
const onChange = (ev) => {
|
|
371
|
+
if (AGENT_BUSY_STATES.has(ev.newState)) return;
|
|
372
|
+
cleanup();
|
|
373
|
+
resolve();
|
|
374
|
+
};
|
|
375
|
+
const cleanup = () => {
|
|
376
|
+
session.off(voice.AgentSessionEventTypes.AgentStateChanged, onChange);
|
|
377
|
+
if (timer) clearTimeout(timer);
|
|
378
|
+
};
|
|
379
|
+
session.on(voice.AgentSessionEventTypes.AgentStateChanged, onChange);
|
|
380
|
+
timer = setTimeout(() => {
|
|
381
|
+
cleanup();
|
|
382
|
+
resolve();
|
|
383
|
+
}, maxWaitMs);
|
|
384
|
+
timer.unref?.();
|
|
385
|
+
});
|
|
337
386
|
}
|
|
387
|
+
/**
|
|
388
|
+
* Ends the call after the agent asked to (via its end-call tool): wait for the agent's closing words
|
|
389
|
+
* to finish, speak an optional final `message`, hold a short drain (`drainMs`) so audio buffered at
|
|
390
|
+
* the caller finishes playing, then disconnect. The teardown's session close
|
|
391
|
+
* force-interrupts any playing speech, so the waits here MUST complete before we disconnect — that's
|
|
392
|
+
* the whole point of the sequence. `ctx.deleteRoom()` hangs up the caller (SIP-safe); `ctx.shutdown()`
|
|
393
|
+
* ends the job and runs the registered shutdown callbacks (`onCallEnd`), so end-of-call work happens
|
|
394
|
+
* exactly as it does on a caller hang-up. Both run in a `finally` so a hiccup while waiting still ends
|
|
395
|
+
* the call. Never rejects — every step is guarded and failures are logged, so callers can safely
|
|
396
|
+
* fire-and-forget it (`void runEndCall(...)`) from hooks. Exported for unit testing.
|
|
397
|
+
*/
|
|
338
398
|
async function runEndCall(session, ctx, config, logger) {
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
logger.warn("@mastra/livekit: shutdown while ending the call failed", error);
|
|
359
|
-
}
|
|
360
|
-
}
|
|
399
|
+
try {
|
|
400
|
+
await waitForAgentDoneSpeaking(session, config.maxWaitMs ?? 3e4);
|
|
401
|
+
if (config.message) await session.say(config.message, { allowInterruptions: false }).waitForPlayout().catch(() => {});
|
|
402
|
+
const drainMs = config.drainMs ?? 800;
|
|
403
|
+
if (drainMs > 0) await new Promise((resolve) => setTimeout(resolve, drainMs));
|
|
404
|
+
} catch (error) {
|
|
405
|
+
logger.warn("@mastra/livekit: waiting for the agent to finish before ending the call failed", error);
|
|
406
|
+
} finally {
|
|
407
|
+
try {
|
|
408
|
+
await ctx.deleteRoom();
|
|
409
|
+
} catch (error) {
|
|
410
|
+
logger.warn("@mastra/livekit: deleteRoom while ending the call failed", error);
|
|
411
|
+
}
|
|
412
|
+
try {
|
|
413
|
+
ctx.shutdown(config.reason ?? "agent ended call");
|
|
414
|
+
} catch (error) {
|
|
415
|
+
logger.warn("@mastra/livekit: shutdown while ending the call failed", error);
|
|
416
|
+
}
|
|
417
|
+
}
|
|
361
418
|
}
|
|
419
|
+
/**
|
|
420
|
+
* Builds the per-turn hook that detects the agent's end-call tool and kicks off the hang-up once, or
|
|
421
|
+
* `undefined` when end-call isn't configured. The detector fires the teardown fire-and-forget (the
|
|
422
|
+
* turn never waits on it) and de-dupes so a repeated tool call can't start two hang-ups.
|
|
423
|
+
*/
|
|
362
424
|
function buildEndCallDetector(config, getSession, ctx, logger) {
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
425
|
+
if (!config) return void 0;
|
|
426
|
+
const toolName = config.tool ?? "endCall";
|
|
427
|
+
let triggered = false;
|
|
428
|
+
return (turnCtx) => {
|
|
429
|
+
if (triggered) return;
|
|
430
|
+
if (!turnCtx.result.toolCalls.some((call) => call.toolName === toolName)) return;
|
|
431
|
+
const session = getSession();
|
|
432
|
+
if (!session) return;
|
|
433
|
+
triggered = true;
|
|
434
|
+
runEndCall(session, ctx, config, logger);
|
|
435
|
+
};
|
|
374
436
|
}
|
|
437
|
+
/** Runs the worker's own turn-complete detector alongside the user's `onTurnComplete`, if any. */
|
|
375
438
|
function composeTurnComplete(user, detector) {
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
439
|
+
if (!detector) return user;
|
|
440
|
+
return (ctx) => {
|
|
441
|
+
detector(ctx);
|
|
442
|
+
return user?.(ctx);
|
|
443
|
+
};
|
|
381
444
|
}
|
|
445
|
+
/**
|
|
446
|
+
* Builds a LiveKit agent worker definition that answers voice sessions with Mastra agents.
|
|
447
|
+
*
|
|
448
|
+
* Use as the default export of your worker entry file, then run it with the LiveKit
|
|
449
|
+
* agents CLI:
|
|
450
|
+
*
|
|
451
|
+
* ```ts
|
|
452
|
+
* // src/mastra/voice-worker.ts
|
|
453
|
+
* import { fileURLToPath } from 'node:url';
|
|
454
|
+
* import { createLiveKitWorker, runLiveKitWorker } from '@mastra/livekit/worker';
|
|
455
|
+
* import { mastra } from './index';
|
|
456
|
+
*
|
|
457
|
+
* export default createLiveKitWorker({
|
|
458
|
+
* mastra,
|
|
459
|
+
* stt: 'deepgram/nova-3',
|
|
460
|
+
* tts: 'cartesia/sonic-3',
|
|
461
|
+
* turnDetection: 'multilingual',
|
|
462
|
+
* });
|
|
463
|
+
*
|
|
464
|
+
* if (process.argv[1] === fileURLToPath(import.meta.url)) {
|
|
465
|
+
* runLiveKitWorker({ entry: import.meta.url, agentName: 'mastra-voice' });
|
|
466
|
+
* }
|
|
467
|
+
* ```
|
|
468
|
+
*/
|
|
382
469
|
function createLiveKitWorker(options) {
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
memory: memoryInstance,
|
|
550
|
-
threadId: memory.thread,
|
|
551
|
-
resourceId: memory.resource ?? memory.thread,
|
|
552
|
-
greeting: greetingText
|
|
553
|
-
});
|
|
554
|
-
} catch (error) {
|
|
555
|
-
logger.warn("@mastra/livekit: failed to persist the greeting", error);
|
|
556
|
-
}
|
|
557
|
-
}
|
|
558
|
-
}
|
|
559
|
-
}
|
|
560
|
-
await options.onSessionStart?.({ session, ctx, agent, metadata });
|
|
561
|
-
} catch (error) {
|
|
562
|
-
voiceObs?.finalize({ error });
|
|
563
|
-
throw error;
|
|
564
|
-
}
|
|
565
|
-
}
|
|
566
|
-
});
|
|
470
|
+
if (options.generate && (options.agent || options.workflow)) throw new Error("@mastra/livekit: set exactly one reply generator — `generate`, `agent`, or `workflow` — not a combination.");
|
|
471
|
+
if (options.agent && options.workflow) throw new Error("@mastra/livekit: set `agent` or `workflow`, not both — they are mutually exclusive reply generators.");
|
|
472
|
+
if (options.workflow && !options.workflowInput) throw new Error("@mastra/livekit: `workflowInput` is required when `workflow` is set. Map the turn into the workflow inputData, e.g. workflowInput: ({ chatCtx }) => ({ history: chatContextToMessages(chatCtx) }).");
|
|
473
|
+
if (options.generate && options.configuration?.endCall) throw new Error("@mastra/livekit: `configuration.endCall` has no effect with `generate` — the worker cannot observe tool calls from a custom reply generator. Detect the end-call tool inside your generator and call `runEndCall` directly instead.");
|
|
474
|
+
const wantsSileroVad = options.vad === void 0 || options.vad === "silero";
|
|
475
|
+
if (options.turnDetection === "multilingual" || options.turnDetection === "english") {
|
|
476
|
+
requestEouMethod(EOU_METHODS[options.turnDetection]);
|
|
477
|
+
queueWorkerSetup(import("@livekit/agents-plugin-livekit").then(() => {
|
|
478
|
+
for (const method of Object.values(EOU_METHODS)) if (!isEouMethodRequested(method)) delete InferenceRunner.registeredRunners[method];
|
|
479
|
+
}).catch(() => {}));
|
|
480
|
+
}
|
|
481
|
+
return defineAgent({
|
|
482
|
+
prewarm: async (proc) => {
|
|
483
|
+
if (wantsSileroVad) proc.userData.vad = await loadSileroVad();
|
|
484
|
+
},
|
|
485
|
+
entry: async (ctx) => {
|
|
486
|
+
const logger = options.mastra.getLogger();
|
|
487
|
+
const metadata = parseSessionMetadata(ctx.job.metadata);
|
|
488
|
+
const args = {
|
|
489
|
+
metadata,
|
|
490
|
+
ctx
|
|
491
|
+
};
|
|
492
|
+
const requestContext = metadata.requestContext ? new RequestContext(Object.entries(metadata.requestContext)) : void 0;
|
|
493
|
+
const endCallDetector = buildEndCallDetector(options.configuration?.endCall, () => session, ctx, logger);
|
|
494
|
+
const onTurnComplete = composeTurnComplete(options.onTurnComplete, endCallDetector);
|
|
495
|
+
let mastraAgent;
|
|
496
|
+
let replyGenerator;
|
|
497
|
+
let agentLabel;
|
|
498
|
+
if (options.generate) {
|
|
499
|
+
replyGenerator = options.generate;
|
|
500
|
+
agentLabel = "mastra-voice";
|
|
501
|
+
} else if (options.workflow) {
|
|
502
|
+
const { workflow } = await resolveWorkflow(options, args);
|
|
503
|
+
agentLabel = workflow.id;
|
|
504
|
+
const mapInput = options.workflowInput;
|
|
505
|
+
replyGenerator = createWorkflowReplyGenerator({
|
|
506
|
+
workflow,
|
|
507
|
+
workflowInput: (turnCtx) => mapInput({
|
|
508
|
+
...turnCtx,
|
|
509
|
+
metadata
|
|
510
|
+
}),
|
|
511
|
+
replyStep: options.replyStep,
|
|
512
|
+
resultText: options.resultText,
|
|
513
|
+
toolFeedback: options.toolFeedback,
|
|
514
|
+
onTurnComplete
|
|
515
|
+
});
|
|
516
|
+
} else {
|
|
517
|
+
mastraAgent = await resolveMastraAgent(options, args);
|
|
518
|
+
agentLabel = mastraAgent.id ?? mastraAgent.name;
|
|
519
|
+
}
|
|
520
|
+
await ctx.connect();
|
|
521
|
+
const roomName = ctx.room.name ?? "mastra-voice";
|
|
522
|
+
let vad;
|
|
523
|
+
if (options.vad && options.vad !== "silero") vad = options.vad;
|
|
524
|
+
else if (wantsSileroVad) vad = ctx.proc.userData.vad ?? await loadSileroVad();
|
|
525
|
+
let turnDetection;
|
|
526
|
+
if (options.turnDetection === "multilingual" || options.turnDetection === "english") turnDetection = await loadTurnDetector(options.turnDetection);
|
|
527
|
+
else turnDetection = options.turnDetection;
|
|
528
|
+
const memory = resolveMemory(options, mastraAgent, args, roomName);
|
|
529
|
+
const memoryInstance = memory ? await resolveMemoryInstance(options, mastraAgent, args, requestContext) : null;
|
|
530
|
+
if (memory && memoryInstance) try {
|
|
531
|
+
await ensureVoiceCallThread({
|
|
532
|
+
memory: memoryInstance,
|
|
533
|
+
threadId: memory.thread,
|
|
534
|
+
resourceId: memory.resource ?? memory.thread,
|
|
535
|
+
roomName
|
|
536
|
+
});
|
|
537
|
+
} catch (error) {
|
|
538
|
+
logger.warn("@mastra/livekit: failed to create the voice call thread", error);
|
|
539
|
+
}
|
|
540
|
+
const voiceObs = options.observability === false ? void 0 : startVoiceCallObservability({
|
|
541
|
+
mastra: options.mastra,
|
|
542
|
+
agentId: agentLabel,
|
|
543
|
+
roomName,
|
|
544
|
+
metadata,
|
|
545
|
+
requestContext
|
|
546
|
+
});
|
|
547
|
+
if (voiceObs) ctx.addShutdownCallback(async () => {
|
|
548
|
+
voiceObs.finalize();
|
|
549
|
+
});
|
|
550
|
+
if (options.onCallEnd) {
|
|
551
|
+
const onCallEnd = options.onCallEnd;
|
|
552
|
+
ctx.addShutdownCallback(async () => {
|
|
553
|
+
try {
|
|
554
|
+
await onCallEnd({
|
|
555
|
+
memory,
|
|
556
|
+
memoryInstance,
|
|
557
|
+
metadata,
|
|
558
|
+
requestContext,
|
|
559
|
+
configuration: options.configuration,
|
|
560
|
+
roomName,
|
|
561
|
+
ctx
|
|
562
|
+
});
|
|
563
|
+
} catch (error) {
|
|
564
|
+
logger.warn("@mastra/livekit: onCallEnd hook threw", error);
|
|
565
|
+
}
|
|
566
|
+
});
|
|
567
|
+
}
|
|
568
|
+
const greetingConfig = resolveGreetingConfig(options);
|
|
569
|
+
const greetingReminder = greetingConfig.repeatEvery && greetingConfig.repeatEvery > 0 ? {
|
|
570
|
+
everyMs: greetingConfig.repeatEvery,
|
|
571
|
+
text: greetingConfig.repeatText
|
|
572
|
+
} : void 0;
|
|
573
|
+
const agent = createMastraVoiceAgent({
|
|
574
|
+
...replyGenerator ? { generate: replyGenerator } : { agent: mastraAgent },
|
|
575
|
+
instructions: mastraAgent ? await resolveInstructions(mastraAgent, requestContext) : void 0,
|
|
576
|
+
memory,
|
|
577
|
+
requestContext,
|
|
578
|
+
toolFeedback: options.toolFeedback,
|
|
579
|
+
onTurnComplete,
|
|
580
|
+
greetingReminder,
|
|
581
|
+
streamOptions: voiceObs ? { tracingContext: voiceObs.tracingContext } : void 0
|
|
582
|
+
});
|
|
583
|
+
const callContext = {
|
|
584
|
+
metadata,
|
|
585
|
+
requestContext,
|
|
586
|
+
roomName,
|
|
587
|
+
ctx
|
|
588
|
+
};
|
|
589
|
+
const [stt, tts] = await Promise.all([resolveSessionComponent(options.configuration?.stt, options.stt, callContext), resolveSessionComponent(options.configuration?.tts, options.tts, callContext)]);
|
|
590
|
+
const session = new voice.AgentSession({
|
|
591
|
+
stt,
|
|
592
|
+
tts,
|
|
593
|
+
vad,
|
|
594
|
+
turnHandling: buildTurnHandling(options, turnDetection),
|
|
595
|
+
...options.sessionOptions
|
|
596
|
+
});
|
|
597
|
+
voiceObs?.attach(session);
|
|
598
|
+
try {
|
|
599
|
+
await session.start({
|
|
600
|
+
agent,
|
|
601
|
+
room: ctx.room,
|
|
602
|
+
inputOptions: options.inputOptions,
|
|
603
|
+
outputOptions: options.outputOptions
|
|
604
|
+
});
|
|
605
|
+
if (greetingConfig.text) {
|
|
606
|
+
const greetingText = await resolveGreetingText(greetingConfig.text, callContext);
|
|
607
|
+
if (greetingText) {
|
|
608
|
+
await speakGreeting(session, {
|
|
609
|
+
...greetingConfig,
|
|
610
|
+
text: greetingText
|
|
611
|
+
});
|
|
612
|
+
if (greetingConfig.persist !== false && memory && memoryInstance) try {
|
|
613
|
+
await persistSpokenGreeting({
|
|
614
|
+
memory: memoryInstance,
|
|
615
|
+
threadId: memory.thread,
|
|
616
|
+
resourceId: memory.resource ?? memory.thread,
|
|
617
|
+
greeting: greetingText
|
|
618
|
+
});
|
|
619
|
+
} catch (error) {
|
|
620
|
+
logger.warn("@mastra/livekit: failed to persist the greeting", error);
|
|
621
|
+
}
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
await options.onSessionStart?.({
|
|
625
|
+
session,
|
|
626
|
+
ctx,
|
|
627
|
+
agent,
|
|
628
|
+
metadata
|
|
629
|
+
});
|
|
630
|
+
} catch (error) {
|
|
631
|
+
voiceObs?.finalize({ error });
|
|
632
|
+
throw error;
|
|
633
|
+
}
|
|
634
|
+
}
|
|
635
|
+
});
|
|
567
636
|
}
|
|
637
|
+
//#endregion
|
|
638
|
+
//#region src/run.ts
|
|
568
639
|
function resolveWorkerEntryPath(entry) {
|
|
569
|
-
|
|
570
|
-
|
|
640
|
+
if (entry instanceof URL) return fileURLToPath(entry);
|
|
641
|
+
return entry.startsWith("file:") ? fileURLToPath(entry) : entry;
|
|
571
642
|
}
|
|
643
|
+
/**
|
|
644
|
+
* Starts the LiveKit agent worker CLI (`dev` / `start` / `connect` subcommands) for a
|
|
645
|
+
* worker entry file. Call it from the same file that default-exports
|
|
646
|
+
* {@link createLiveKitWorker}, guarded so it only runs when executed directly:
|
|
647
|
+
*
|
|
648
|
+
* ```ts
|
|
649
|
+
* import { fileURLToPath } from 'node:url';
|
|
650
|
+
* import { createLiveKitWorker, runLiveKitWorker } from '@mastra/livekit/worker';
|
|
651
|
+
* import { mastra } from './index';
|
|
652
|
+
*
|
|
653
|
+
* export default createLiveKitWorker({ mastra, agent: 'support' });
|
|
654
|
+
*
|
|
655
|
+
* if (process.argv[1] === fileURLToPath(import.meta.url)) {
|
|
656
|
+
* runLiveKitWorker({ entry: import.meta.url, agentName: 'mastra-voice' });
|
|
657
|
+
* }
|
|
658
|
+
* ```
|
|
659
|
+
*
|
|
660
|
+
* Using this helper (instead of `cli.runApp` from `@livekit/agents`) guarantees the worker
|
|
661
|
+
* runtime and the bridge share one copy of the LiveKit SDK.
|
|
662
|
+
*/
|
|
572
663
|
function runLiveKitWorker(options) {
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
);
|
|
581
|
-
});
|
|
664
|
+
workerSetupComplete().then(() => {
|
|
665
|
+
cli.runApp(new ServerOptions({
|
|
666
|
+
agent: resolveWorkerEntryPath(options.entry),
|
|
667
|
+
agentName: options.agentName ?? "mastra-voice",
|
|
668
|
+
...options.serverOptions
|
|
669
|
+
}));
|
|
670
|
+
});
|
|
582
671
|
}
|
|
672
|
+
//#endregion
|
|
673
|
+
export { DEFAULT_END_CALL_DRAIN_MS, DEFAULT_END_CALL_MAX_WAIT_MS, DEFAULT_END_CALL_REASON, DEFAULT_END_CALL_TOOL, chatContextToMessages, createLiveKitWorker, createRemoteAgentReplyGenerator, runEndCall, runLiveKitWorker, speakGreeting, waitForAgentDoneSpeaking };
|
|
583
674
|
|
|
584
|
-
export { DEFAULT_END_CALL_DRAIN_MS, DEFAULT_END_CALL_MAX_WAIT_MS, DEFAULT_END_CALL_REASON, DEFAULT_END_CALL_TOOL, createLiveKitWorker, runEndCall, runLiveKitWorker, speakGreeting, waitForAgentDoneSpeaking };
|
|
585
|
-
//# sourceMappingURL=worker-entry.js.map
|
|
586
675
|
//# sourceMappingURL=worker-entry.js.map
|