@mastra/livekit 0.0.0 → 0.2.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/CHANGELOG.md +80 -0
  2. package/LICENSE.md +30 -0
  3. package/README.md +270 -7
  4. package/dist/bridge.d.ts +153 -0
  5. package/dist/bridge.d.ts.map +1 -0
  6. package/dist/chunk-2E3MTAOA.js +133 -0
  7. package/dist/chunk-2E3MTAOA.js.map +1 -0
  8. package/dist/chunk-MWTEZOBS.cjs +139 -0
  9. package/dist/chunk-MWTEZOBS.cjs.map +1 -0
  10. package/dist/constants.d.ts +3 -0
  11. package/dist/constants.d.ts.map +1 -0
  12. package/dist/dispatch.d.ts +20 -0
  13. package/dist/dispatch.d.ts.map +1 -0
  14. package/dist/index.cjs +105 -0
  15. package/dist/index.cjs.map +1 -0
  16. package/dist/index.d.ts +10 -0
  17. package/dist/index.d.ts.map +1 -0
  18. package/dist/index.js +91 -0
  19. package/dist/index.js.map +1 -0
  20. package/dist/messages.d.ts +24 -0
  21. package/dist/messages.d.ts.map +1 -0
  22. package/dist/metadata.d.ts +17 -0
  23. package/dist/metadata.d.ts.map +1 -0
  24. package/dist/observability.d.ts +46 -0
  25. package/dist/observability.d.ts.map +1 -0
  26. package/dist/routes.d.ts +52 -0
  27. package/dist/routes.d.ts.map +1 -0
  28. package/dist/run.d.ts +32 -0
  29. package/dist/run.d.ts.map +1 -0
  30. package/dist/voice-thread.d.ts +23 -0
  31. package/dist/voice-thread.d.ts.map +1 -0
  32. package/dist/worker-entry.cjs +656 -0
  33. package/dist/worker-entry.cjs.map +1 -0
  34. package/dist/worker-entry.d.ts +8 -0
  35. package/dist/worker-entry.d.ts.map +1 -0
  36. package/dist/worker-entry.js +652 -0
  37. package/dist/worker-entry.js.map +1 -0
  38. package/dist/worker-setup.d.ts +5 -0
  39. package/dist/worker-setup.d.ts.map +1 -0
  40. package/dist/worker.d.ts +187 -0
  41. package/dist/worker.d.ts.map +1 -0
  42. package/dist/workflow-generator.d.ts +92 -0
  43. package/dist/workflow-generator.d.ts.map +1 -0
  44. package/package.json +24 -14
package/CHANGELOG.md ADDED
@@ -0,0 +1,80 @@
1
+ # @mastra/livekit
2
+
3
+ ## 0.2.0-alpha.0
4
+
5
+ ### Minor Changes
6
+
7
+ - Added `@mastra/livekit`, a new package that turns Mastra agents into realtime voice agents using LiveKit. ([#17896](https://github.com/mastra-ai/mastra/pull/17896))
8
+
9
+ LiveKit's agents framework runs the audio loop — WebRTC transport, voice activity detection, streaming speech-to-text, semantic turn detection, and barge-in — while your Mastra agent generates every reply with its own model, tools, and memory. When a caller interrupts the agent, LiveKit cancels the in-flight stream and Mastra stops generating.
10
+
11
+ **Build a voice worker**
12
+ - `createLiveKitWorker()` builds a LiveKit worker that answers voice sessions with your Mastra agents; `runLiveKitWorker()` starts its CLI (`dev`/`start`). Both live on the `@mastra/livekit/worker` entry point.
13
+ - `liveKitConnectionRoute()` is an API route that mints LiveKit tokens and dispatches the voice agent into a room; `dispatchVoiceSession()` does the same programmatically for server-initiated sessions like outbound calls. These live on the `@mastra/livekit` entry point, which is safe to import from Mastra server code — it never loads the LiveKit agents runtime.
14
+
15
+ ```ts
16
+ // src/mastra/voice-worker.ts
17
+ import { createLiveKitWorker } from '@mastra/livekit/worker';
18
+ import { mastra } from './index';
19
+
20
+ export default createLiveKitWorker({
21
+ mastra,
22
+ agent: 'support',
23
+ stt: 'deepgram/nova-3',
24
+ tts: 'cartesia/sonic-3',
25
+ turnDetection: 'multilingual',
26
+ });
27
+ ```
28
+
29
+ **Drive replies with an agent or a workflow**
30
+
31
+ Each turn's reply can come from a Mastra agent (the default) or a Mastra workflow. With a workflow, LiveKit still owns the audio loop and calls into Mastra once per turn, so the workflow runs to completion each turn (no suspend/resume) — pass the transcript in, stream the reply out.
32
+ - `workflow` / `workflowInput` options on `createLiveKitWorker()` drive replies with a workflow.
33
+ - `pipeAgentReplyToWriter(agentStream, writer)` streams an agent's reply from inside a workflow step, forwarding both its words and its tool calls (piping only the text would drop the tool calls).
34
+ - `generate` is an escape hatch to plug in any custom reply generator.
35
+
36
+ ```ts
37
+ export default createLiveKitWorker({
38
+ mastra,
39
+ workflow: 'phoneConversation',
40
+ workflowInput: ({ messages }) => ({ turn: messages }),
41
+ replyStep: 'generateResponse',
42
+ stt: 'deepgram/nova-3',
43
+ tts: 'cartesia/sonic-3',
44
+ });
45
+ ```
46
+
47
+ **Run work after each turn and at the end of the call**
48
+ - `onTurnComplete` runs once per turn, right after the reply finishes playing. It runs in the background — the worker never waits for it — so you can save memory, update your CRM, or record analytics without adding any delay for the caller or the next reply. It also runs with `result.interrupted: true` when the caller talks over the agent.
49
+ - `onCallEnd` runs once when the call ends. Unlike `onTurnComplete`, the worker waits for it to finish before exiting, so it's the place for end-of-call work like summarizing the whole conversation into long-term memory once.
50
+ - `toolFeedback` speaks a short phrase while a tool runs; `memoryInstance` gives the workflow path a `Memory` instance to open the call's thread and save the greeting, so the saved conversation is complete — greeting included — like the agent path.
51
+
52
+ Both hooks work whether you drive replies with an agent or a workflow.
53
+
54
+ ```ts
55
+ createLiveKitWorker({
56
+ mastra,
57
+ agent: 'callCenter',
58
+ onTurnComplete: async ({ result, memory }) => {
59
+ if (memory) await crm.logContact(memory.resource, result.text);
60
+ },
61
+ onCallEnd: async ({ memory }) => {
62
+ // After the caller hangs up: save a lasting summary of the call.
63
+ },
64
+ });
65
+ ```
66
+
67
+ **Built-in observability**
68
+
69
+ When the Mastra instance has observability configured, each call opens a `voice call` trace that nests every turn's agent run and adds child spans for LiveKit's speech-to-text, text-to-speech, turn-detection, and LLM latency, closing with a per-model token, character, and audio usage roll-up. On by default; pass `observability: false` to disable.
70
+
71
+ **Studio voice mode**
72
+
73
+ Studio's agent chat gains a voice call mode: when the Mastra server exposes a LiveKit connection route and a voice worker is running, a phone button in the chat composer starts a realtime voice session with the agent. Live captions, agent state (listening, thinking, speaking), and barge-in all surface in the chat, and the conversation lands in the same memory thread as text chat.
74
+
75
+ See the [LiveKit voice guide](https://mastra.ai/docs/voice/livekit) for setup.
76
+
77
+ ### Patch Changes
78
+
79
+ - Updated dependencies [[`a0085fa`](https://github.com/mastra-ai/mastra/commit/a0085fa0934e52c37c8c8b3d75a6bb5cd199af36)]:
80
+ - @mastra/core@1.50.0-alpha.5
package/LICENSE.md ADDED
@@ -0,0 +1,30 @@
1
+ Portions of this software are licensed as follows:
2
+
3
+ - All content that resides under any directory named "ee/" within this
4
+ repository, including but not limited to:
5
+ - `packages/core/src/auth/ee/`
6
+ - `packages/server/src/server/auth/ee/`
7
+ is licensed under the license defined in `ee/LICENSE`.
8
+
9
+ - All third-party components incorporated into the Mastra Software are
10
+ licensed under the original license provided by the owner of the
11
+ applicable component.
12
+
13
+ - Content outside of the above-mentioned directories or restrictions is
14
+ available under the "Apache License 2.0" as defined below.
15
+
16
+ # Apache License 2.0
17
+
18
+ Copyright (c) 2025 Kepler Software, Inc.
19
+
20
+ Licensed under the Apache License, Version 2.0 (the "License");
21
+ you may not use this file except in compliance with the License.
22
+ You may obtain a copy of the License at
23
+
24
+ http://www.apache.org/licenses/LICENSE-2.0
25
+
26
+ Unless required by applicable law or agreed to in writing, software
27
+ distributed under the License is distributed on an "AS IS" BASIS,
28
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
29
+ See the License for the specific language governing permissions and
30
+ limitations under the License.
package/README.md CHANGED
@@ -1,6 +1,22 @@
1
1
  # @mastra/livekit
2
2
 
3
- LiveKit voice integration for Mastra agents. LiveKit's agents framework runs the audio loop — WebRTC transport, voice activity detection, streaming speech-to-text, semantic turn detection, barge-in, and text-to-speech — and this package bridges reply generation to a Mastra agent's `stream()` call, so tools, memory, and model routing all run inside Mastra.
3
+ Realtime voice for [Mastra](https://mastra.ai) agents and workflows, powered by [LiveKit Agents](https://docs.livekit.io/agents/).
4
+
5
+ LiveKit's agents framework owns the **audio loop** — WebRTC transport, voice activity detection (VAD), streaming speech-to-text (STT), semantic turn detection, barge-in, and text-to-speech (TTS). This package bridges **reply generation** to Mastra, so each detected user turn is answered by a Mastra **agent** (`agent.stream()`) or **workflow** — with your tools, memory, processors, and model routing all running inside Mastra.
6
+
7
+ ```
8
+ caller speaks ─▶ VAD ─▶ STT ─▶ turn detection ─▶ [ Mastra agent / workflow ] ─▶ TTS ─▶ caller hears
9
+ (LiveKit owns the audio loop) (this package bridges replies)
10
+ ```
11
+
12
+ ## What's in the box
13
+
14
+ - **Two reply paths** — answer turns with a Mastra **agent** (the default, richest path) or a Mastra **workflow** (run-to-completion per turn, e.g. deterministic intent routing). A low-level `generate` escape hatch accepts any custom reply generator.
15
+ - **Full speech stack, pluggable** — STT/TTS as LiveKit inference model strings (`'deepgram/nova-3'`, `'cartesia/sonic-3'`) or your own plugin instances; Silero VAD and LiveKit multilingual/English turn detection; barge-in cancels in-flight generation automatically.
16
+ - **Memory, scoped to the call** — `thread` = call, `resource` = caller, so a returning caller is recognized across calls. Up-front thread creation and greeting persistence keep the saved thread a faithful transcript. Works on the agent path and the workflow path (via `memoryInstance`).
17
+ - **Lifecycle hooks** — `toolFeedback` (speak filler while a tool runs), `onTurnComplete` (post-turn, fire-and-forget, off the audio path), and `onCallEnd` (end-of-call, awaited within LiveKit's shutdown window — the place to flush observational memory).
18
+ - **Observability** — one `voice call` trace per session with LiveKit pipeline metrics and every Mastra run nested under it.
19
+ - **Connection + dispatch helpers** — `liveKitConnectionRoute` mints tokens and dispatches the worker so a frontend can join.
4
20
 
5
21
  ## Installation
6
22
 
@@ -8,22 +24,37 @@ LiveKit voice integration for Mastra agents. LiveKit's agents framework runs the
8
24
  npm install @mastra/livekit @livekit/agents @livekit/agents-plugin-silero @livekit/agents-plugin-livekit
9
25
  ```
10
26
 
11
- ## Usage
27
+ Peer dependencies (`@mastra/core` and `@livekit/agents` are required; the two plugins are optional but enable the defaults):
12
28
 
13
- Create a worker entry file:
29
+ | Package | Needed for |
30
+ | -------------------------------- | -------------------------------------------- |
31
+ | `@mastra/core` | the Mastra agent/workflow you bridge to |
32
+ | `@livekit/agents` | the audio loop runtime |
33
+ | `@livekit/agents-plugin-silero` | the default `vad: 'silero'` |
34
+ | `@livekit/agents-plugin-livekit` | `turnDetection: 'multilingual' \| 'english'` |
35
+
36
+ The package has two entry points, one per process:
37
+
38
+ - `@mastra/livekit` — server-side helpers (`liveKitConnectionRoute`, `dispatchVoiceSession`, `pipeAgentReplyToWriter`). Safe to import from Mastra server code; never loads the `@livekit/agents` runtime.
39
+ - `@mastra/livekit/worker` — the worker runtime (`createLiveKitWorker`, `runLiveKitWorker`). Import it only from the worker entry file.
40
+
41
+ ## Quick start
42
+
43
+ A worker is a standalone Node process that connects to LiveKit and answers sessions. Define it with `createLiveKitWorker` and run it with `runLiveKitWorker`:
14
44
 
15
45
  ```typescript
16
46
  // src/mastra/voice-worker.ts
17
47
  import { fileURLToPath } from 'node:url';
18
- import { createLiveKitWorker, runLiveKitWorker } from '@mastra/livekit';
48
+ import { createLiveKitWorker, runLiveKitWorker } from '@mastra/livekit/worker';
19
49
  import { mastra } from './index';
20
50
 
21
51
  export default createLiveKitWorker({
22
52
  mastra,
23
- agent: 'support',
53
+ agent: 'support', // a Mastra agent key/id (or a resolver, or use `workflow` instead)
24
54
  stt: 'deepgram/nova-3',
25
55
  tts: 'cartesia/sonic-3',
26
56
  turnDetection: 'multilingual',
57
+ greeting: 'Thanks for calling. How can I help?',
27
58
  });
28
59
 
29
60
  if (process.argv[1] === fileURLToPath(import.meta.url)) {
@@ -31,7 +62,7 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
31
62
  }
32
63
  ```
33
64
 
34
- Add a connection endpoint so frontends can join voice sessions:
65
+ Add a connection endpoint to your Mastra server so a frontend can join, then dispatch the worker:
35
66
 
36
67
  ```typescript
37
68
  // src/mastra/index.ts
@@ -48,10 +79,242 @@ export const mastra = new Mastra({
48
79
  Run the worker alongside your Mastra server:
49
80
 
50
81
  ```bash
51
- npx livekit-agents download-files
82
+ npx livekit-agents download-files # one-time: turn-detection + VAD model files
52
83
  npx tsx src/mastra/voice-worker.ts dev
53
84
  ```
54
85
 
86
+ The model strings (`deepgram/nova-3`, `cartesia/sonic-3`) route through **LiveKit Cloud inference**, so with a LiveKit Cloud project you don't need separate Deepgram/Cartesia accounts — only your `LIVEKIT_URL` / `LIVEKIT_API_KEY` / `LIVEKIT_API_SECRET` (plus whatever key your Mastra model needs). To bring your own providers, pass plugin instances to `stt` / `tts` instead of strings.
87
+
88
+ ## Reply paths
89
+
90
+ ### Agent (default)
91
+
92
+ Pass `agent` (a key/id, an `Agent` instance, or a resolver). The agent runs its full loop each turn — model, tools, memory, processors — and streams its text deltas to TTS. Barge-in cancels the in-flight `agent.stream()`.
93
+
94
+ ### Workflow
95
+
96
+ Pass `workflow` + `workflowInput` instead of `agent` (mutually exclusive). LiveKit owns the turn boundary, so the workflow runs **once to completion per turn** — no suspend/resume. Use it for deterministic per-turn structure (e.g. classify intent, then reply).
97
+
98
+ ```typescript
99
+ import { createLiveKitWorker, chatContextToMessages } from '@mastra/livekit/worker';
100
+
101
+ export default createLiveKitWorker({
102
+ mastra,
103
+ workflow: 'phoneConversation',
104
+ workflowInput: ({ messages, memory }) => ({ turn: messages, memory: memory || undefined }),
105
+ replyStep: 'generateResponse', // only stream text from this step (optional)
106
+ stt: 'deepgram/nova-3',
107
+ tts: 'cartesia/sonic-3',
108
+ turnDetection: 'multilingual',
109
+ });
110
+ ```
111
+
112
+ In the reply-producing step, use **`pipeAgentReplyToWriter`** to forward the agent's reply into the step `writer`. It streams text deltas (so TTS starts early) **and** tool-call chunks (so `toolFeedback` fires and `onTurnComplete` sees the tool list) — unlike piping only `.textStream`, which silently drops tool calls:
113
+
114
+ ```typescript
115
+ import { pipeAgentReplyToWriter } from '@mastra/livekit';
116
+
117
+ const generateResponse = createStep({
118
+ id: 'generateResponse',
119
+ execute: async ({ inputData, mastra, writer, abortSignal }) => {
120
+ const stream = await mastra.getAgent('support').stream(inputData.turn, {
121
+ memory: inputData.memory, // engages working memory, recall, etc.
122
+ abortSignal, // lets barge-in stop generation promptly
123
+ });
124
+ const reply = await pipeAgentReplyToWriter(stream, writer);
125
+ return { reply };
126
+ },
127
+ });
128
+ ```
129
+
130
+ A step that writes no text stays silent unless you pass `resultText` to derive the reply from the final run result.
131
+
132
+ ### Custom (`generate`)
133
+
134
+ For full control, pass a `generate` function — any `VoiceReplyGenerator` that turns a turn into a `ReadableStream<string>` (a remote bridge, a bespoke pipeline, …).
135
+
136
+ ## `createLiveKitWorker` options
137
+
138
+ | Option | Type | Notes |
139
+ | --------------------------------------------------- | ----------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
140
+ | `mastra` | `Mastra` | **Required.** The instance whose agents/workflows answer sessions. |
141
+ | **Reply generation** (pick one) | | |
142
+ | `agent` | `string \| Agent \| (args) => …` | The agent that answers. Defaults to `metadata.agentId`. |
143
+ | `workflow` | `string \| Workflow \| (args) => string` | Answer with a workflow instead. Requires `workflowInput`. |
144
+ | `workflowInput` | `(ctx & { metadata }) => inputData` | Maps a turn into the workflow's `inputData`. |
145
+ | `replyStep` | `string` | Only stream text from this workflow step id. |
146
+ | `resultText` | `(result) => string` | Fallback reply text when the workflow streams nothing. |
147
+ | `generate` | `VoiceReplyGenerator` | Lowest-level escape hatch. |
148
+ | **Speech stack** | | |
149
+ | `stt` | plugin or `'provider/model'` | Speech-to-text. |
150
+ | `tts` | plugin or `'provider/model'` | Text-to-speech. |
151
+ | `vad` | `VAD \| 'silero' \| false` | Voice activity detection. Defaults to `'silero'`. |
152
+ | `turnDetection` | `'multilingual' \| 'english' \| …` | End-of-turn detection. |
153
+ | `turnHandling` | `AgentSessionOptions['turnHandling']` | Endpointing delays, interruption sensitivity, preemptive generation. |
154
+ | `sessionOptions` / `inputOptions` / `outputOptions` | partial LiveKit options | Merged over what the helper builds. |
155
+ | **Memory** | | |
156
+ | `memory` | `false \| (args) => { thread, resource }` | Memory mapping. Defaults to `{ thread: metadata.threadId ?? room, resource: metadata.resourceId ?? thread }` when the agent has memory. |
157
+ | `memoryInstance` | `Memory \| (args) => Memory` | The `Memory` used to bootstrap the thread + persist the greeting on the **workflow/custom** path (no agent to source it from). Mastra storage is injected if the `Memory` has none. |
158
+ | `greeting` | `string` | Spoken when the session starts. |
159
+ | `persistGreeting` | `boolean` | Save the greeting to the thread. Defaults to `true`. |
160
+ | **Lifecycle hooks** | | |
161
+ | `toolFeedback` | `(toolCall) => string \| void` | Speak filler while a tool runs (agent + workflow). |
162
+ | `onTurnComplete` | `VoiceTurnCompleteHook` | After each turn streams, **fire-and-forget**, off the audio path. |
163
+ | `onCallEnd` | `VoiceCallEndHook` | When the call ends, **awaited** within LiveKit's shutdown window. |
164
+ | `onSessionStart` | `(args) => …` | After the session starts — attach listeners, trigger replies, etc. |
165
+ | **Other** | | |
166
+ | `observability` | `boolean` | Voice-pipeline tracing. Defaults to `true`. |
167
+
168
+ ## Lifecycle hooks
169
+
170
+ Three hooks let you do work around a turn without adding to the caller's latency:
171
+
172
+ ```typescript
173
+ createLiveKitWorker({
174
+ mastra,
175
+ agent: 'support',
176
+
177
+ // 1. In-turn: speak a short phrase while a tool runs, so the caller isn't left in silence.
178
+ toolFeedback: ({ toolName }) => (toolName === 'lookupOrder' ? 'Let me pull that up.' : undefined),
179
+
180
+ // 2. Post-turn: fire-and-forget AFTER the reply has streamed — the worker never awaits it, so it
181
+ // can't delay the caller or the next turn. Carries the produced reply + the memory mapping.
182
+ onTurnComplete: async ({ result, memory }) => {
183
+ if (memory) await crm.logContact(memory.resource, result.text); // result.text/toolCalls/interrupted
184
+ },
185
+
186
+ // 3. End-of-call: runs when the caller hangs up, AWAITED within LiveKit's shutdown grace window
187
+ // (so it finishes before the process exits). The place for end-of-call memory maintenance.
188
+ onCallEnd: async ({ memory, memoryInstance }) => {
189
+ // e.g. flush observational memory once for the whole call instead of paying for it per turn.
190
+ },
191
+ });
192
+ ```
193
+
194
+ `onTurnComplete` and `toolFeedback` work on the **workflow** path too (the reply step must surface tool calls via `pipeAgentReplyToWriter`).
195
+
196
+ ## Joining a call: `liveKitConnectionRoute`
197
+
198
+ Mounts an API route on your Mastra server that mints a LiveKit token and dispatches the worker by `agentName`. Frontends `POST` to it to get connection details.
199
+
200
+ | Option | Default | Notes |
201
+ | ------------------------------------ | -------------------------------------------------------- | --------------------------------------------------- |
202
+ | `path` | `/voice/livekit/connection-details` | Must not start with `/api`. |
203
+ | `serverUrl` / `apiKey` / `apiSecret` | `LIVEKIT_URL` / `LIVEKIT_API_KEY` / `LIVEKIT_API_SECRET` | LiveKit credentials. |
204
+ | `agentName` | — | Must match the worker's `agentName`. |
205
+ | `ttl` | `'15m'` | Token lifetime. |
206
+ | `requiresAuth` | `true` | Mastra custom routes require auth unless opted out. |
207
+ | `roomName` / `participantIdentity` | generated | String or `(args) => string`. |
208
+ | `metadata` | passes `agentId`/`threadId`/`resourceId` | Session metadata delivered to the worker. |
209
+
210
+ For programmatic dispatch (no HTTP), use `dispatchVoiceSession`.
211
+
212
+ ## Running the worker: `runLiveKitWorker`
213
+
214
+ Starts the LiveKit agent worker CLI (`dev` / `start` / `connect`) for your entry file.
215
+
216
+ | Option | Default | Notes |
217
+ | --------------- | ---------------- | --------------------------------------------------- |
218
+ | `entry` | — | The worker module; pass `import.meta.url`. |
219
+ | `agentName` | `'mastra-voice'` | Dispatch name; must match `liveKitConnectionRoute`. |
220
+ | `serverOptions` | — | Extra LiveKit `ServerOptions`. |
221
+
222
+ ## Observability
223
+
224
+ When the Mastra instance has observability configured, the worker opens one `voice call` span per session, nests every turn's Mastra run under it, and adds child spans for LiveKit pipeline metrics — STT, TTS, end-of-utterance, VAD, and LLM time-to-first-token — closing with a per-model token/character/audio usage roll-up. On by default; pass `observability: false` to disable.
225
+
226
+ ## Deployment
227
+
228
+ A voice deployment is **two long-running processes** plus a LiveKit media server:
229
+
230
+ | Component | What runs it | Notes |
231
+ | ------------------------ | ------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- |
232
+ | **LiveKit media server** | LiveKit Cloud, or self-hosted `livekit-server` + Redis | WebRTC transport. Both processes below need its `LIVEKIT_URL` / `LIVEKIT_API_KEY` / `LIVEKIT_API_SECRET`. |
233
+ | **Mastra HTTP server** | `mastra build` → `node .mastra/output/index.mjs` | Your agents/workflows + `liveKitConnectionRoute` (mints tokens, dispatches the worker). |
234
+ | **LiveKit worker** | your worker entry → `runLiveKitWorker` (LiveKit Agents CLI `start`) | Connects **outbound** to the media server and answers calls. Not an HTTP server. |
235
+
236
+ The worker is a **separate process**: `mastra build` bundles only your `Mastra` instance and `src/mastra/tools/**` — never the worker entry, because nothing imports it. `liveKitConnectionRoute`, by contrast, is an `apiRoute`, so it ships inside the server build automatically. The server and worker therefore build and run independently and can live on different hosts, as long as they share a LiveKit project and the same `agentName`.
237
+
238
+ > Note: `mastra worker build` is unrelated — it bundles Mastra's own pubsub/scheduler workflow workers (`mastra.startWorkers()`), not this LiveKit worker.
239
+
240
+ ### Building the worker
241
+
242
+ The worker imports `@livekit/agents` (plus the optional plugins) and your `Mastra` instance, then runs the LiveKit Agents CLI. Two build-time essentials:
243
+
244
+ - **Run `start`, not `dev`, in production** — `dev` is hot-reload only.
245
+ - **Pre-download the model files** so they're baked into the image instead of fetched on cold start:
246
+
247
+ ```bash
248
+ node --import tsx src/mastra/voice-worker.ts download-files # Silero VAD + turn-detector ONNX
249
+ ```
250
+
251
+ Running the worker with `tsx` against source avoids bundling LiveKit's native deps (onnxruntime, …). If you do, make `tsx` and the `@livekit/agents*` packages **real** dependencies (not devDependencies) in the deployed image.
252
+
253
+ ### Docker
254
+
255
+ Build the server with `mastra build`, bake the model files, then run the two processes. **Recommended: one image, two services** — workers scale by call volume and the HTTP server scales by request volume, so keep them independent:
256
+
257
+ ```dockerfile
258
+ FROM node:22-slim AS build
259
+ WORKDIR /app
260
+ COPY package.json package-lock.json ./
261
+ RUN npm ci
262
+ COPY . .
263
+ RUN npx mastra build
264
+ RUN node --import tsx src/mastra/voice-worker.ts download-files
265
+
266
+ FROM node:22-slim
267
+ WORKDIR /app
268
+ COPY --from=build /app /app
269
+ ENV NODE_ENV=production
270
+ EXPOSE 4111
271
+ # server service: CMD ["node", ".mastra/output/index.mjs"]
272
+ # worker service: CMD ["node", "--import", "tsx", "src/mastra/voice-worker.ts", "start"]
273
+ ```
274
+
275
+ To run **both in one container** (single-tenant boxes, demos), supervise them with `bash` so the container exits — and is restarted by the orchestrator — if either dies:
276
+
277
+ ```bash
278
+ #!/usr/bin/env bash
279
+ set -euo pipefail
280
+ node .mastra/output/index.mjs &
281
+ node --import tsx src/mastra/voice-worker.ts start &
282
+ wait -n
283
+ exit 1
284
+ ```
285
+
286
+ This is simpler, but it couples two processes that have opposite scaling curves and no independent autoscaling — prefer the split for anything beyond a demo.
287
+
288
+ ### Managed platforms (Mastra Cloud, Railway, Cloud Run, …)
289
+
290
+ Single-process HTTP hosts run the **Mastra server** as-is (`node .mastra/output/index.mjs`). They can't host the worker — it isn't an HTTP server, it isn't in the build output, and it's a long-lived outbound connection. Deploy the **hybrid**: the server on the managed platform (it still mints tokens and dispatches via the bundled connection route), and the worker on any plain process host (a dedicated Railway/Fly/Render service, a VM, a Kubernetes `Deployment`, ECS, …). Point both at the same LiveKit project, keep ≥1 always-on worker instance (no scale-to-zero, so it stays registered), and inject the shared `LIVEKIT_*` env into both.
291
+
292
+ ## Runnable example
293
+
294
+ A complete, runnable reference lives in the Mastra monorepo at **[`examples/voice-agent`](https://github.com/mastra-ai/mastra/tree/main/examples/voice-agent)** — a trades-contractor front-desk voice agent that exercises nearly every feature here:
295
+
296
+ - **Both entrypoints** — an agent worker (`pnpm worker`) and a workflow worker (`pnpm worker:workflow`, deterministic intent routing → memory-backed reply).
297
+ - **Three memory layers** — working memory, semantic recall, and observational memory, all scoped to the caller.
298
+ - **Tools + deterministic reconciliation**, a tenant-context input processor, the `toolFeedback` / `onTurnComplete` / `onCallEnd` hooks, and full observability.
299
+
300
+ To run it:
301
+
302
+ ```bash
303
+ git clone https://github.com/mastra-ai/mastra
304
+ cd mastra && pnpm install && pnpm build:packages # build the workspace packages
305
+
306
+ cd examples/voice-agent
307
+ cp .env.example .env # add LiveKit Cloud creds + your model key (e.g. OPENAI_API_KEY)
308
+ pnpm install
309
+ pnpm worker:download-files # one-time model download
310
+
311
+ # in two terminals:
312
+ pnpm dev # Mastra server + Studio at http://localhost:4111
313
+ pnpm worker # the voice worker (or `pnpm worker:workflow`)
314
+ ```
315
+
316
+ See the example's own [`README.md`](https://github.com/mastra-ai/mastra/tree/main/examples/voice-agent) for the scenario walkthrough and the memory/latency design notes.
317
+
55
318
  ## Documentation
56
319
 
57
320
  - [Using LiveKit with Mastra](https://mastra.ai/docs/voice/livekit)
@@ -0,0 +1,153 @@
1
+ import { ReadableStream } from 'node:stream/web';
2
+ import { llm, voice } from '@livekit/agents';
3
+ import type { Agent as MastraAgent, AgentExecutionOptionsBase } from '@mastra/core/agent';
4
+ import type { TracingContext } from '@mastra/core/observability';
5
+ import { RequestContext } from '@mastra/core/request-context';
6
+ import type { VoiceTurnMessage } from './messages.js';
7
+ export type MastraStreamOptions = Partial<AgentExecutionOptionsBase<unknown>>;
8
+ export interface VoiceToolCall {
9
+ toolCallId: string;
10
+ toolName: string;
11
+ args?: unknown;
12
+ }
13
+ export interface MastraVoiceAgentMemory {
14
+ thread: string;
15
+ resource?: string;
16
+ }
17
+ /**
18
+ * Per-turn context handed to a {@link VoiceReplyGenerator}. LiveKit calls `llmNode` once per
19
+ * detected user turn; the bridge builds this context and asks the generator for the reply.
20
+ */
21
+ export interface VoiceTurnContext {
22
+ /**
23
+ * The messages to generate a reply from. With Mastra Memory on, only the messages new since
24
+ * the agent last spoke (history comes from the thread); with memory off, the full session.
25
+ *
26
+ * For a workflow / custom generator: pass these straight to a memory-backed `agent.stream(...,
27
+ * { memory })` inside a step so the agent backfills history from the thread (no duplication). A
28
+ * stateless workflow that wants the entire transcript every turn should read `chatCtx` instead
29
+ * (e.g. `chatContextToMessages(chatCtx)`), since there is no thread to backfill from.
30
+ */
31
+ messages: VoiceTurnMessage[];
32
+ /** The raw LiveKit chat context, for generators that want the full transcript or message parts. */
33
+ chatCtx: llm.ChatContext;
34
+ /** Resolved memory mapping for the call, or `false` when memory is disabled. */
35
+ memory: MastraVoiceAgentMemory | false;
36
+ /** Request context forwarded to generation. */
37
+ requestContext?: RequestContext;
38
+ /** Voice-call span context, so each turn's generation nests under the call trace. */
39
+ tracingContext?: TracingContext;
40
+ }
41
+ /**
42
+ * What a turn produced, handed to {@link VoiceTurnCompleteHook} after the reply finishes.
43
+ */
44
+ export interface VoiceTurnResult {
45
+ /** The assistant reply text streamed this turn, accumulated from the model's text deltas. */
46
+ text: string;
47
+ /** Tool calls the agent made during the turn, in order. */
48
+ toolCalls: VoiceToolCall[];
49
+ /** True when barge-in cut the turn short before it finished streaming. */
50
+ interrupted: boolean;
51
+ }
52
+ /** {@link VoiceTurnContext} plus the reply it produced. Passed to {@link VoiceTurnCompleteHook}. */
53
+ export interface VoiceTurnCompleteContext extends VoiceTurnContext {
54
+ /** The reply the agent produced this turn. */
55
+ result: VoiceTurnResult;
56
+ }
57
+ /**
58
+ * Called once per turn AFTER the reply has finished streaming to text-to-speech — off the audio
59
+ * path. It runs fire-and-forget: the turn does not await it, so post-turn work (memory
60
+ * maintenance, CRM writes, analytics) never delays what the caller hears or the next turn. A
61
+ * thrown error or rejected promise is logged, not propagated. Because the resolved `memory`
62
+ * mapping (`thread`/`resource`) is on the context, this is the place for a truly non-blocking
63
+ * `memory.updateWorkingMemory(...)`. See {@link MastraVoiceAgentOptions.onTurnComplete}.
64
+ */
65
+ export type VoiceTurnCompleteHook = (ctx: VoiceTurnCompleteContext) => void | Promise<void>;
66
+ /**
67
+ * Produces a stream of text deltas for one conversational turn, or `null` to stay silent.
68
+ * Cancelling the returned stream (LiveKit does this on barge-in) must abort the underlying
69
+ * generation. Built-in implementations: {@link createAgentReplyGenerator} (a Mastra agent) and
70
+ * `createWorkflowReplyGenerator` (a Mastra workflow).
71
+ */
72
+ export type VoiceReplyGenerator = (ctx: VoiceTurnContext) => ReadableStream<string> | null | Promise<ReadableStream<string> | null>;
73
+ export interface AgentReplyGeneratorOptions {
74
+ /** The Mastra agent that generates replies. Tools and memory run inside this agent. */
75
+ agent: MastraAgent;
76
+ /** Extra options merged into every `agent.stream()` call (e.g. `tracingContext`). */
77
+ streamOptions?: MastraStreamOptions;
78
+ /** Speak a short phrase while a tool call runs. See {@link MastraVoiceAgentOptions.toolFeedback}. */
79
+ toolFeedback?: (toolCall: VoiceToolCall) => string | undefined | void;
80
+ /** Fired off the audio path after the reply streams. See {@link MastraVoiceAgentOptions.onTurnComplete}. */
81
+ onTurnComplete?: VoiceTurnCompleteHook;
82
+ }
83
+ /**
84
+ * A {@link VoiceReplyGenerator} backed by a Mastra agent: runs the agent's full loop (model,
85
+ * tools, memory) and streams its text deltas. On barge-in the returned stream is cancelled,
86
+ * which aborts the in-flight `agent.stream()`.
87
+ */
88
+ export declare function createAgentReplyGenerator(options: AgentReplyGeneratorOptions): VoiceReplyGenerator;
89
+ export interface MastraVoiceAgentOptions {
90
+ /**
91
+ * The Mastra agent that generates replies. Tools and memory run inside this agent. Provide
92
+ * either this or {@link MastraVoiceAgentOptions.generate}.
93
+ */
94
+ agent?: MastraAgent;
95
+ /**
96
+ * A lower-level reply generator (e.g. from `createWorkflowReplyGenerator`). Use instead of
97
+ * `agent` to drive replies with a workflow or any custom generator.
98
+ */
99
+ generate?: VoiceReplyGenerator;
100
+ /**
101
+ * Conversation persistence. When set, only messages new since the agent last spoke are
102
+ * sent each turn and Mastra Memory supplies history. When `false`, the full LiveKit
103
+ * in-session context is sent on every turn instead.
104
+ */
105
+ memory?: MastraVoiceAgentMemory | false;
106
+ /** Request context entries forwarded to generation. */
107
+ requestContext?: RequestContext | Record<string, unknown>;
108
+ /**
109
+ * Called when the Mastra agent starts a tool call mid-reply. Return a short phrase (e.g. "Let
110
+ * me look that up.") to speak it while the tool runs; it also appears in the transcript. Return
111
+ * nothing to stay silent. Applies to the agent generator built here; the workflow generator
112
+ * takes its own equivalent via `createWorkflowReplyGenerator`.
113
+ */
114
+ toolFeedback?: (toolCall: VoiceToolCall) => string | undefined | void;
115
+ /**
116
+ * Called once per turn after the reply has finished streaming to text-to-speech. Runs off the
117
+ * audio path and fire-and-forget — the turn does not await it — so post-turn memory
118
+ * maintenance, CRM writes, or analytics never delay the caller or the next turn. The context
119
+ * carries the produced reply ({@link VoiceTurnResult}) and the resolved `memory` mapping, so
120
+ * this is where a truly non-blocking `memory.updateWorkingMemory(...)` belongs. A thrown error
121
+ * or rejected promise is logged, not propagated. Applies to the agent generator built here; the
122
+ * workflow generator takes its own via `createWorkflowReplyGenerator`.
123
+ */
124
+ onTurnComplete?: VoiceTurnCompleteHook;
125
+ /** Extra options merged into every `agent.stream()` call (agent generator only). */
126
+ streamOptions?: MastraStreamOptions;
127
+ /** LiveKit agent instructions. Unused for reply generation (the Mastra agent/workflow applies its own). */
128
+ instructions?: string;
129
+ id?: voice.AgentOptions<unknown>['id'];
130
+ stt?: voice.AgentOptions<unknown>['stt'];
131
+ vad?: voice.AgentOptions<unknown>['vad'];
132
+ tts?: voice.AgentOptions<unknown>['tts'];
133
+ turnHandling?: voice.AgentOptions<unknown>['turnHandling'];
134
+ }
135
+ /**
136
+ * A LiveKit `voice.Agent` whose replies come from a Mastra agent or workflow.
137
+ *
138
+ * LiveKit keeps ownership of the audio loop (VAD, STT, turn detection, TTS, barge-in) and calls
139
+ * `llmNode` once per detected user turn; the node delegates to a {@link VoiceReplyGenerator}
140
+ * which streams text deltas back. On barge-in LiveKit cancels the returned stream, which aborts
141
+ * the in-flight generation.
142
+ */
143
+ export declare class MastraVoiceAgent extends voice.Agent {
144
+ readonly mastraAgent?: MastraAgent;
145
+ readonly memory: MastraVoiceAgentMemory | false;
146
+ readonly requestContext?: RequestContext;
147
+ readonly streamOptions?: MastraStreamOptions;
148
+ private readonly replyGenerator;
149
+ constructor(options: MastraVoiceAgentOptions);
150
+ llmNode(chatCtx: llm.ChatContext, _toolCtx: llm.ToolContext, _modelSettings: voice.ModelSettings): Promise<ReadableStream<llm.ChatChunk | string> | null>;
151
+ }
152
+ export declare function createMastraVoiceAgent(options: MastraVoiceAgentOptions): MastraVoiceAgent;
153
+ //# sourceMappingURL=bridge.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"bridge.d.ts","sourceRoot":"","sources":["../src/bridge.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,iBAAiB,CAAC;AACjD,OAAO,EAAE,GAAG,EAAE,KAAK,EAAE,MAAM,iBAAiB,CAAC;AAC7C,OAAO,KAAK,EAAE,KAAK,IAAI,WAAW,EAAE,yBAAyB,EAAE,MAAM,oBAAoB,CAAC;AAC1F,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAC;AACjE,OAAO,EAAE,cAAc,EAAE,MAAM,8BAA8B,CAAC;AAE9D,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,YAAY,CAAC;AAInD,MAAM,MAAM,mBAAmB,GAAG,OAAO,CAAC,yBAAyB,CAAC,OAAO,CAAC,CAAC,CAAC;AAE9E,MAAM,WAAW,aAAa;IAC5B,UAAU,EAAE,MAAM,CAAC;IACnB,QAAQ,EAAE,MAAM,CAAC;IACjB,IAAI,CAAC,EAAE,OAAO,CAAC;CAChB;AAED,MAAM,WAAW,sBAAsB;IACrC,MAAM,EAAE,MAAM,CAAC;IACf,QAAQ,CAAC,EAAE,MAAM,CAAC;CACnB;AAED;;;GAGG;AACH,MAAM,WAAW,gBAAgB;IAC/B;;;;;;;;OAQG;IACH,QAAQ,EAAE,gBAAgB,EAAE,CAAC;IAC7B,mGAAmG;IACnG,OAAO,EAAE,GAAG,CAAC,WAAW,CAAC;IACzB,gFAAgF;IAChF,MAAM,EAAE,sBAAsB,GAAG,KAAK,CAAC;IACvC,+CAA+C;IAC/C,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,qFAAqF;IACrF,cAAc,CAAC,EAAE,cAAc,CAAC;CACjC;AAED;;GAEG;AACH,MAAM,WAAW,eAAe;IAC9B,6FAA6F;IAC7F,IAAI,EAAE,MAAM,CAAC;IACb,2DAA2D;IAC3D,SAAS,EAAE,aAAa,EAAE,CAAC;IAC3B,0EAA0E;IAC1E,WAAW,EAAE,OAAO,CAAC;CACtB;AAED,oGAAoG;AACpG,MAAM,WAAW,wBAAyB,SAAQ,gBAAgB;IAChE,8CAA8C;IAC9C,MAAM,EAAE,eAAe,CAAC;CACzB;AAED;;;;;;;GAOG;AACH,MAAM,MAAM,qBAAqB,GAAG,CAAC,GAAG,EAAE,wBAAwB,KAAK,IAAI,GAAG,OAAO,CAAC,IAAI,CAAC,CAAC;AAE5F;;;;;GAKG;AACH,MAAM,MAAM,mBAAmB,GAAG,CAChC,GAAG,EAAE,gBAAgB,KAClB,cAAc,CAAC,MAAM,CAAC,GAAG,IAAI,GAAG,OAAO,CAAC,cAAc,CAAC,MAAM,CAAC,GAAG,IAAI,CAAC,CAAC;AAE5E,MAAM,WAAW,0BAA0B;IACzC,uFAAuF;IACvF,KAAK,EAAE,WAAW,CAAC;IACnB,qFAAqF;IACrF,aAAa,CAAC,EAAE,mBAAmB,CAAC;IACpC,qGAAqG;IACrG,YAAY,CAAC,EAAE,CAAC,QAAQ,EAAE,aAAa,KAAK,MAAM,GAAG,SAAS,GAAG,IAAI,CAAC;IACtE,4GAA4G;IAC5G,cAAc,CAAC,EAAE,qBAAqB,CAAC;CACxC;AAED;;;;GAIG;AACH,wBAAgB,yBAAyB,CAAC,OAAO,EAAE,0BAA0B,GAAG,mBAAmB,CA4ElG;AAED,MAAM,WAAW,uBAAuB;IACtC;;;OAGG;IACH,KAAK,CAAC,EAAE,WAAW,CAAC;IACpB;;;OAGG;IACH,QAAQ,CAAC,EAAE,mBAAmB,CAAC;IAC/B;;;;OAIG;IACH,MAAM,CAAC,EAAE,sBAAsB,GAAG,KAAK,CAAC;IACxC,uDAAuD;IACvD,cAAc,CAAC,EAAE,cAAc,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAC1D;;;;;OAKG;IACH,YAAY,CAAC,EAAE,CAAC,QAAQ,EAAE,aAAa,KAAK,MAAM,GAAG,SAAS,GAAG,IAAI,CAAC;IACtE;;;;;;;;OAQG;IACH,cAAc,CAAC,EAAE,qBAAqB,CAAC;IACvC,oFAAoF;IACpF,aAAa,CAAC,EAAE,mBAAmB,CAAC;IACpC,2GAA2G;IAC3G,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,EAAE,CAAC,EAAE,KAAK,CAAC,YAAY,CAAC,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC;IACvC,GAAG,CAAC,EAAE,KAAK,CAAC,YAAY,CAAC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;IACzC,GAAG,CAAC,EAAE,KAAK,CAAC,YAAY,CAAC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;IACzC,GAAG,CAAC,EAAE,KAAK,CAAC,YAAY,CAAC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC;IACzC,YAAY,CAAC,EAAE,KAAK,CAAC,YAAY,CAAC,OAAO,CAAC,CAAC,cAAc,CAAC,CAAC;CAC5D;AAiCD;;;;;;;GAOG;AACH,qBAAa,gBAAiB,SAAQ,KAAK,CAAC,KAAK;IAC/C,QAAQ,CAAC,WAAW,CAAC,EAAE,WAAW,CAAC;IACnC,QAAQ,CAAC,MAAM,EAAE,sBAAsB,GAAG,KAAK,CAAC;IAChD,QAAQ,CAAC,cAAc,CAAC,EAAE,cAAc,CAAC;IACzC,QAAQ,CAAC,aAAa,CAAC,EAAE,mBAAmB,CAAC;IAC7C,OAAO,CAAC,QAAQ,CAAC,cAAc,CAAsB;gBAEzC,OAAO,EAAE,uBAAuB;IAkC7B,OAAO,CACpB,OAAO,EAAE,GAAG,CAAC,WAAW,EACxB,QAAQ,EAAE,GAAG,CAAC,WAAW,EACzB,cAAc,EAAE,KAAK,CAAC,aAAa,GAClC,OAAO,CAAC,cAAc,CAAC,GAAG,CAAC,SAAS,GAAG,MAAM,CAAC,GAAG,IAAI,CAAC;CAa1D;AAED,wBAAgB,sBAAsB,CAAC,OAAO,EAAE,uBAAuB,GAAG,gBAAgB,CAEzF"}