@mastra/livekit 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,201 +1,12 @@
1
1
  import { parseSessionMetadata, createWorkflowReplyGenerator, DEFAULT_LIVEKIT_AGENT_NAME } from './chunk-2E3MTAOA.js';
2
- import { InferenceRunner, defineAgent, voice, cli, ServerOptions, metrics, llm } from '@livekit/agents';
2
+ import { createMastraVoiceAgent } from './chunk-4O7IN74Y.js';
3
+ export { chatContextToMessages, createRemoteAgentReplyGenerator } from './chunk-4O7IN74Y.js';
4
+ import { voice, InferenceRunner, defineAgent, cli, ServerOptions, metrics } from '@livekit/agents';
3
5
  import { RequestContext } from '@mastra/core/request-context';
4
- import { ReadableStream } from 'stream/web';
5
6
  import { getOrCreateSpan, SpanType } from '@mastra/core/observability';
6
7
  import { randomUUID } from 'crypto';
7
8
  import { fileURLToPath } from 'url';
8
9
 
9
- // src/messages.ts
10
- function textOfMessage(message) {
11
- const parts = [];
12
- for (const part of message.content) {
13
- if (typeof part === "string") {
14
- parts.push(part);
15
- } else if (part.type === "instructions") {
16
- parts.push(part.value);
17
- } else if (part.type === "audio_content" && part.transcript) {
18
- parts.push(part.transcript);
19
- }
20
- }
21
- return parts.join("\n").trim();
22
- }
23
- function toVoiceTurnMessage(item) {
24
- if (item.type !== "message") return void 0;
25
- const content = textOfMessage(item);
26
- if (!content) return void 0;
27
- if (item.role === "user") return { role: "user", content };
28
- if (item.role === "assistant") return { role: "assistant", content };
29
- return { role: "system", content };
30
- }
31
- function extractNewTurnMessages(chatCtx) {
32
- const items = chatCtx.items;
33
- let lastAssistantIdx = -1;
34
- for (let i = items.length - 1; i >= 0; i--) {
35
- const item = items[i];
36
- if (item?.type === "message" && item.role === "assistant") {
37
- lastAssistantIdx = i;
38
- break;
39
- }
40
- }
41
- const messages = [];
42
- for (const item of items.slice(lastAssistantIdx + 1)) {
43
- const message = toVoiceTurnMessage(item);
44
- if (message) messages.push(message);
45
- }
46
- return messages;
47
- }
48
- function chatContextToMessages(chatCtx) {
49
- const withoutInstructions = chatCtx.copy({ excludeInstructions: true, excludeFunctionCall: true });
50
- const messages = [];
51
- for (const item of withoutInstructions.items) {
52
- const message = toVoiceTurnMessage(item);
53
- if (message) messages.push(message);
54
- }
55
- return messages;
56
- }
57
-
58
- // src/bridge.ts
59
- var DEFAULT_INSTRUCTIONS = "You are a helpful voice assistant powered by a Mastra agent.";
60
- function createAgentReplyGenerator(options) {
61
- const { agent, streamOptions, toolFeedback, onTurnComplete } = options;
62
- return (ctx) => {
63
- if (ctx.messages.length === 0) return null;
64
- const abortController = new AbortController();
65
- const mergedOptions = {
66
- ...streamOptions,
67
- abortSignal: abortController.signal
68
- };
69
- if (ctx.memory) mergedOptions.memory = ctx.memory;
70
- if (ctx.requestContext) mergedOptions.requestContext = ctx.requestContext;
71
- let cancelled = false;
72
- let replyText = "";
73
- const toolCalls = [];
74
- const emitTurnComplete = (interrupted) => {
75
- if (!onTurnComplete) return;
76
- const completeCtx = { ...ctx, result: { text: replyText, toolCalls, interrupted } };
77
- Promise.resolve().then(() => onTurnComplete(completeCtx)).catch((error) => {
78
- console.warn("@mastra/livekit: onTurnComplete hook threw", error);
79
- });
80
- };
81
- return new ReadableStream({
82
- start: async (controller) => {
83
- try {
84
- const result = await agent.stream(ctx.messages, mergedOptions);
85
- for await (const chunk of result.fullStream) {
86
- if (cancelled) break;
87
- if (chunk.type === "text-delta") {
88
- if (chunk.payload.text) {
89
- replyText += chunk.payload.text;
90
- controller.enqueue(chunk.payload.text);
91
- }
92
- } else if (chunk.type === "tool-call") {
93
- const toolCall = {
94
- toolCallId: chunk.payload.toolCallId,
95
- toolName: chunk.payload.toolName,
96
- args: chunk.payload.args
97
- };
98
- toolCalls.push(toolCall);
99
- if (toolFeedback) {
100
- const filler = toolFeedback(toolCall);
101
- if (filler) controller.enqueue(filler.endsWith(" ") ? filler : `${filler} `);
102
- }
103
- } else if (chunk.type === "error") {
104
- const error = chunk.payload.error;
105
- throw error instanceof Error ? error : new Error(String(error));
106
- }
107
- }
108
- if (!cancelled) controller.close();
109
- emitTurnComplete(cancelled);
110
- } catch (error) {
111
- if (cancelled || abortController.signal.aborted) {
112
- emitTurnComplete(true);
113
- return;
114
- }
115
- controller.error(error);
116
- }
117
- },
118
- cancel: () => {
119
- cancelled = true;
120
- abortController.abort();
121
- }
122
- });
123
- };
124
- }
125
- function toRequestContext(value) {
126
- if (!value) return void 0;
127
- if (value instanceof RequestContext) return value;
128
- return new RequestContext(Object.entries(value));
129
- }
130
- var MastraPlaceholderLLM = class extends llm.LLM {
131
- label() {
132
- return "mastra.MastraVoiceAgent";
133
- }
134
- get model() {
135
- return "mastra-agent";
136
- }
137
- get provider() {
138
- return "mastra";
139
- }
140
- chat() {
141
- throw new Error(
142
- "@mastra/livekit: reply generation runs through the Mastra agent via llmNode; the placeholder LLM cannot be used for inference."
143
- );
144
- }
145
- };
146
- var MastraVoiceAgent = class extends voice.Agent {
147
- mastraAgent;
148
- memory;
149
- requestContext;
150
- streamOptions;
151
- replyGenerator;
152
- constructor(options) {
153
- if (options.agent && options.generate) {
154
- throw new Error(
155
- "@mastra/livekit: MastraVoiceAgent requires `agent` or `generate`, not both \u2014 they are mutually exclusive reply sources."
156
- );
157
- }
158
- super({
159
- id: options.id,
160
- instructions: options.instructions ?? DEFAULT_INSTRUCTIONS,
161
- stt: options.stt,
162
- vad: options.vad,
163
- llm: new MastraPlaceholderLLM(),
164
- tts: options.tts,
165
- turnHandling: options.turnHandling
166
- });
167
- this.memory = options.memory ?? false;
168
- this.requestContext = toRequestContext(options.requestContext);
169
- this.streamOptions = options.streamOptions;
170
- if (options.generate) {
171
- this.replyGenerator = options.generate;
172
- } else if (options.agent) {
173
- this.mastraAgent = options.agent;
174
- this.replyGenerator = createAgentReplyGenerator({
175
- agent: options.agent,
176
- streamOptions: options.streamOptions,
177
- toolFeedback: options.toolFeedback,
178
- onTurnComplete: options.onTurnComplete
179
- });
180
- } else {
181
- throw new Error("@mastra/livekit: MastraVoiceAgent requires `agent` or `generate`.");
182
- }
183
- }
184
- async llmNode(chatCtx, _toolCtx, _modelSettings) {
185
- const messages = this.memory === false ? chatContextToMessages(chatCtx) : extractNewTurnMessages(chatCtx);
186
- if (messages.length === 0) return null;
187
- return this.replyGenerator({
188
- messages,
189
- chatCtx,
190
- memory: this.memory,
191
- requestContext: this.requestContext,
192
- tracingContext: this.streamOptions?.tracingContext
193
- });
194
- }
195
- };
196
- function createMastraVoiceAgent(options) {
197
- return new MastraVoiceAgent(options);
198
- }
199
10
  function modelMeta(metadata) {
200
11
  const out = {};
201
12
  if (metadata?.modelProvider) out.modelProvider = metadata.modelProvider;
@@ -471,6 +282,103 @@ async function resolveInstructions(mastraAgent, requestContext) {
471
282
  return void 0;
472
283
  }
473
284
  }
285
+ function resolveGreetingConfig(options) {
286
+ return {
287
+ text: options.greeting,
288
+ persist: options.persistGreeting,
289
+ ...options.configuration?.greeting
290
+ };
291
+ }
292
+ async function resolveGreetingText(text, context) {
293
+ const resolved = typeof text === "function" ? await text(context) : text;
294
+ return resolved || void 0;
295
+ }
296
+ async function resolveSessionComponent(resolver, fallback, context) {
297
+ const resolved = resolver ? await resolver(context) : void 0;
298
+ return resolved ?? fallback;
299
+ }
300
+ async function speakGreeting(session, greeting) {
301
+ if (!greeting.text) return void 0;
302
+ const handle = session.say(
303
+ greeting.text,
304
+ greeting.allowInterruptions === void 0 ? void 0 : { allowInterruptions: greeting.allowInterruptions }
305
+ );
306
+ if (greeting.awaitPlayout) {
307
+ await handle.waitForPlayout().catch(() => {
308
+ });
309
+ }
310
+ return handle;
311
+ }
312
+ var DEFAULT_END_CALL_TOOL = "endCall";
313
+ var DEFAULT_END_CALL_REASON = "agent ended call";
314
+ var DEFAULT_END_CALL_MAX_WAIT_MS = 3e4;
315
+ var DEFAULT_END_CALL_DRAIN_MS = 800;
316
+ var AGENT_BUSY_STATES = /* @__PURE__ */ new Set(["thinking", "speaking"]);
317
+ function waitForAgentDoneSpeaking(session, maxWaitMs = DEFAULT_END_CALL_MAX_WAIT_MS) {
318
+ if (!AGENT_BUSY_STATES.has(session.agentState)) return Promise.resolve();
319
+ return new Promise((resolve) => {
320
+ let timer;
321
+ const onChange = (ev) => {
322
+ if (AGENT_BUSY_STATES.has(ev.newState)) return;
323
+ cleanup();
324
+ resolve();
325
+ };
326
+ const cleanup = () => {
327
+ session.off(voice.AgentSessionEventTypes.AgentStateChanged, onChange);
328
+ if (timer) clearTimeout(timer);
329
+ };
330
+ session.on(voice.AgentSessionEventTypes.AgentStateChanged, onChange);
331
+ timer = setTimeout(() => {
332
+ cleanup();
333
+ resolve();
334
+ }, maxWaitMs);
335
+ timer.unref?.();
336
+ });
337
+ }
338
+ async function runEndCall(session, ctx, config, logger) {
339
+ try {
340
+ await waitForAgentDoneSpeaking(session, config.maxWaitMs ?? DEFAULT_END_CALL_MAX_WAIT_MS);
341
+ if (config.message) {
342
+ await session.say(config.message, { allowInterruptions: false }).waitForPlayout().catch(() => {
343
+ });
344
+ }
345
+ const drainMs = config.drainMs ?? DEFAULT_END_CALL_DRAIN_MS;
346
+ if (drainMs > 0) await new Promise((resolve) => setTimeout(resolve, drainMs));
347
+ } catch (error) {
348
+ logger.warn("@mastra/livekit: waiting for the agent to finish before ending the call failed", error);
349
+ } finally {
350
+ try {
351
+ await ctx.deleteRoom();
352
+ } catch (error) {
353
+ logger.warn("@mastra/livekit: deleteRoom while ending the call failed", error);
354
+ }
355
+ try {
356
+ ctx.shutdown(config.reason ?? DEFAULT_END_CALL_REASON);
357
+ } catch (error) {
358
+ logger.warn("@mastra/livekit: shutdown while ending the call failed", error);
359
+ }
360
+ }
361
+ }
362
+ function buildEndCallDetector(config, getSession, ctx, logger) {
363
+ if (!config) return void 0;
364
+ const toolName = config.tool ?? DEFAULT_END_CALL_TOOL;
365
+ let triggered = false;
366
+ return (turnCtx) => {
367
+ if (triggered) return;
368
+ if (!turnCtx.result.toolCalls.some((call) => call.toolName === toolName)) return;
369
+ const session = getSession();
370
+ if (!session) return;
371
+ triggered = true;
372
+ void runEndCall(session, ctx, config, logger);
373
+ };
374
+ }
375
+ function composeTurnComplete(user, detector) {
376
+ if (!detector) return user;
377
+ return (ctx) => {
378
+ void detector(ctx);
379
+ return user?.(ctx);
380
+ };
381
+ }
474
382
  function createLiveKitWorker(options) {
475
383
  if (options.generate && (options.agent || options.workflow)) {
476
384
  throw new Error(
@@ -487,6 +395,11 @@ function createLiveKitWorker(options) {
487
395
  "@mastra/livekit: `workflowInput` is required when `workflow` is set. Map the turn into the workflow inputData, e.g. workflowInput: ({ chatCtx }) => ({ history: chatContextToMessages(chatCtx) })."
488
396
  );
489
397
  }
398
+ if (options.generate && options.configuration?.endCall) {
399
+ throw new Error(
400
+ "@mastra/livekit: `configuration.endCall` has no effect with `generate` \u2014 the worker cannot observe tool calls from a custom reply generator. Detect the end-call tool inside your generator and call `runEndCall` directly instead."
401
+ );
402
+ }
490
403
  const wantsSileroVad = options.vad === void 0 || options.vad === "silero";
491
404
  if (options.turnDetection === "multilingual" || options.turnDetection === "english") {
492
405
  requestEouMethod(EOU_METHODS[options.turnDetection]);
@@ -512,6 +425,8 @@ function createLiveKitWorker(options) {
512
425
  const metadata = parseSessionMetadata(ctx.job.metadata);
513
426
  const args = { metadata, ctx };
514
427
  const requestContext = metadata.requestContext ? new RequestContext(Object.entries(metadata.requestContext)) : void 0;
428
+ const endCallDetector = buildEndCallDetector(options.configuration?.endCall, () => session, ctx, logger);
429
+ const onTurnComplete = composeTurnComplete(options.onTurnComplete, endCallDetector);
515
430
  let mastraAgent;
516
431
  let replyGenerator;
517
432
  let agentLabel;
@@ -528,7 +443,7 @@ function createLiveKitWorker(options) {
528
443
  replyStep: options.replyStep,
529
444
  resultText: options.resultText,
530
445
  toolFeedback: options.toolFeedback,
531
- onTurnComplete: options.onTurnComplete
446
+ onTurnComplete
532
447
  });
533
448
  } else {
534
449
  mastraAgent = await resolveMastraAgent(options, args);
@@ -578,24 +493,40 @@ function createLiveKitWorker(options) {
578
493
  const onCallEnd = options.onCallEnd;
579
494
  ctx.addShutdownCallback(async () => {
580
495
  try {
581
- await onCallEnd({ memory, memoryInstance, metadata, requestContext, roomName, ctx });
496
+ await onCallEnd({
497
+ memory,
498
+ memoryInstance,
499
+ metadata,
500
+ requestContext,
501
+ configuration: options.configuration,
502
+ roomName,
503
+ ctx
504
+ });
582
505
  } catch (error) {
583
506
  logger.warn("@mastra/livekit: onCallEnd hook threw", error);
584
507
  }
585
508
  });
586
509
  }
510
+ const greetingConfig = resolveGreetingConfig(options);
511
+ const greetingReminder = greetingConfig.repeatEvery && greetingConfig.repeatEvery > 0 ? { everyMs: greetingConfig.repeatEvery, text: greetingConfig.repeatText } : void 0;
587
512
  const agent = createMastraVoiceAgent({
588
513
  ...replyGenerator ? { generate: replyGenerator } : { agent: mastraAgent },
589
514
  instructions: mastraAgent ? await resolveInstructions(mastraAgent, requestContext) : void 0,
590
515
  memory,
591
516
  requestContext,
592
517
  toolFeedback: options.toolFeedback,
593
- onTurnComplete: options.onTurnComplete,
518
+ onTurnComplete,
519
+ greetingReminder,
594
520
  streamOptions: voiceObs ? { tracingContext: voiceObs.tracingContext } : void 0
595
521
  });
522
+ const callContext = { metadata, requestContext, roomName, ctx };
523
+ const [stt, tts] = await Promise.all([
524
+ resolveSessionComponent(options.configuration?.stt, options.stt, callContext),
525
+ resolveSessionComponent(options.configuration?.tts, options.tts, callContext)
526
+ ]);
596
527
  const session = new voice.AgentSession({
597
- stt: options.stt,
598
- tts: options.tts,
528
+ stt,
529
+ tts,
599
530
  vad,
600
531
  turnHandling: buildTurnHandling(options, turnDetection),
601
532
  ...options.sessionOptions
@@ -608,18 +539,21 @@ function createLiveKitWorker(options) {
608
539
  inputOptions: options.inputOptions,
609
540
  outputOptions: options.outputOptions
610
541
  });
611
- if (options.greeting) {
612
- session.say(options.greeting);
613
- if (options.persistGreeting !== false && memory && memoryInstance) {
614
- try {
615
- await persistSpokenGreeting({
616
- memory: memoryInstance,
617
- threadId: memory.thread,
618
- resourceId: memory.resource ?? memory.thread,
619
- greeting: options.greeting
620
- });
621
- } catch (error) {
622
- logger.warn("@mastra/livekit: failed to persist the greeting", error);
542
+ if (greetingConfig.text) {
543
+ const greetingText = await resolveGreetingText(greetingConfig.text, callContext);
544
+ if (greetingText) {
545
+ await speakGreeting(session, { ...greetingConfig, text: greetingText });
546
+ if (greetingConfig.persist !== false && memory && memoryInstance) {
547
+ try {
548
+ await persistSpokenGreeting({
549
+ memory: memoryInstance,
550
+ threadId: memory.thread,
551
+ resourceId: memory.resource ?? memory.thread,
552
+ greeting: greetingText
553
+ });
554
+ } catch (error) {
555
+ logger.warn("@mastra/livekit: failed to persist the greeting", error);
556
+ }
623
557
  }
624
558
  }
625
559
  }
@@ -647,6 +581,6 @@ function runLiveKitWorker(options) {
647
581
  });
648
582
  }
649
583
 
650
- export { chatContextToMessages, createLiveKitWorker, runLiveKitWorker };
584
+ export { DEFAULT_END_CALL_DRAIN_MS, DEFAULT_END_CALL_MAX_WAIT_MS, DEFAULT_END_CALL_REASON, DEFAULT_END_CALL_TOOL, createLiveKitWorker, runEndCall, runLiveKitWorker, speakGreeting, waitForAgentDoneSpeaking };
651
585
  //# sourceMappingURL=worker-entry.js.map
652
586
  //# sourceMappingURL=worker-entry.js.map