@mastra/mcp-docs-server 1.2.26-alpha.1 → 1.2.26-alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/.docs/docs/agents/overview.md +1 -1
  2. package/.docs/docs/agents/processors.md +21 -0
  3. package/.docs/docs/deployment/mastra-server.md +8 -2
  4. package/.docs/docs/evals/datasets.md +5 -1
  5. package/.docs/docs/guides/context-engineering.md +1 -1
  6. package/.docs/docs/harness/background-tasks.md +30 -24
  7. package/.docs/docs/harness/signals.md +39 -0
  8. package/.docs/docs/index.md +1 -1
  9. package/.docs/docs/memory/message-history.md +6 -2
  10. package/.docs/docs/subagents.md +25 -0
  11. package/.docs/integrations/agentic-ui/ai-sdk-ui.md +7 -0
  12. package/.docs/integrations/file-storage/amazon-s3.md +7 -1
  13. package/.docs/integrations/file-storage/archil.md +3 -3
  14. package/.docs/integrations/frameworks/electron.md +1 -1
  15. package/.docs/integrations/sandboxes/cloudflare-sandbox.md +2 -0
  16. package/.docs/integrations/sandboxes/daytona.md +33 -0
  17. package/.docs/integrations/sandboxes/docker.md +13 -0
  18. package/.docs/integrations/voice/livekit.md +26 -2
  19. package/.docs/models/gateways/merge-gateway.md +6 -1
  20. package/.docs/models/gateways/netlify.md +2 -1
  21. package/.docs/models/gateways/openrouter.md +8 -2
  22. package/.docs/models/index.md +1 -1
  23. package/.docs/models/providers/above.md +1 -1
  24. package/.docs/models/providers/baseten.md +2 -1
  25. package/.docs/models/providers/bothub.md +2 -1
  26. package/.docs/models/providers/cline-pass.md +18 -17
  27. package/.docs/models/providers/cortecs.md +3 -3
  28. package/.docs/models/providers/deepinfra.md +6 -6
  29. package/.docs/models/providers/edenai.md +18 -4
  30. package/.docs/models/providers/empiriolabs.md +3 -1
  31. package/.docs/models/providers/fireworks-ai.md +2 -1
  32. package/.docs/models/providers/greenpt.md +2 -1
  33. package/.docs/models/providers/huggingface.md +4 -1
  34. package/.docs/models/providers/hyper.md +6 -5
  35. package/.docs/models/providers/kilo.md +15 -10
  36. package/.docs/models/providers/llmgateway-providers.md +5 -2
  37. package/.docs/models/providers/llmgateway.md +3 -1
  38. package/.docs/models/providers/nan.md +1 -1
  39. package/.docs/models/providers/nano-gpt.md +13 -4
  40. package/.docs/models/providers/ofox.md +30 -3
  41. package/.docs/models/providers/ollama-cloud.md +2 -1
  42. package/.docs/models/providers/pioneer.md +11 -2
  43. package/.docs/models/providers/requesty.md +2 -2
  44. package/.docs/models/providers/volcengine-coding-plan.md +3 -1
  45. package/.docs/reference/agents/agent.md +16 -0
  46. package/.docs/reference/agents/generate.md +2 -0
  47. package/.docs/reference/client-js/datasets.md +1 -1
  48. package/.docs/reference/configuration.md +2 -2
  49. package/.docs/reference/datasets/purgeItem.md +3 -3
  50. package/.docs/reference/index.md +3 -0
  51. package/.docs/reference/memory/cloneThread.md +2 -0
  52. package/.docs/reference/memory/copyThread.md +65 -0
  53. package/.docs/reference/memory/memory-class.md +2 -1
  54. package/.docs/reference/memory/recall.md +51 -0
  55. package/.docs/reference/memory/updateThreadResourceId.md +46 -0
  56. package/.docs/reference/processors/agents-md-injector.md +55 -0
  57. package/.docs/reference/processors/processor-interface.md +2 -0
  58. package/.docs/reference/pubsub/redis-streams.md +6 -0
  59. package/.docs/reference/pubsub/valkey-streams.md +6 -0
  60. package/.docs/reference/streaming/agents/stream.md +30 -0
  61. package/.docs/reference/workspace/filesystem.md +72 -0
  62. package/package.json +5 -5
@@ -76,7 +76,7 @@ Short list of known model IDs are:
76
76
 
77
77
  Go to <https://mastra.ai/models> for a full list of supported models.
78
78
 
79
- Add a tool an agent by importing the tool and passing it to the agent constructor as a tools object.
79
+ Add a tool to an agent by importing the tool and passing it to the agent constructor as a tools object.
80
80
 
81
81
  Example:
82
82
 
@@ -976,6 +976,27 @@ export class ContextLengthHandler implements Processor {
976
976
 
977
977
  Mastra includes a built-in [`PrefillErrorHandler`](https://mastra.ai/reference/processors/prefill-error-handler) that automatically handles the Anthropic "assistant message prefill" error. This processor is auto-injected and requires no configuration.
978
978
 
979
+ ## Receive background work notifications
980
+
981
+ Use `createBackgroundWorkSignalProcessor()` to retain the active caller's signal capability for eligible background tool calls:
982
+
983
+ ```typescript
984
+ import { Agent } from '@mastra/core/agent'
985
+ import { createBackgroundWorkSignalProcessor } from '@mastra/core/processors'
986
+
987
+ const agent = new Agent({
988
+ id: 'background-agent',
989
+ name: 'Background agent',
990
+ instructions: 'Complete tasks using the available tools.',
991
+ model: 'openai/gpt-5.6-sol',
992
+ inputProcessors: [createBackgroundWorkSignalProcessor()],
993
+ })
994
+ ```
995
+
996
+ Mastra's background runtime remains responsible for execution, persistence, retries, result reconciliation, and continuation. The processor only retains caller-scoped notification access. When background work starts, it emits `work-deferred` or `work-awaited` with `status: 'running'`. After the authoritative tool result is reconciled, it emits `work-completed` or `work-failed`.
997
+
998
+ Notifications are process-local and best-effort. If the originating run has already ended or the signal can't be delivered (for example, because the task resumed on another process), Mastra preserves the authoritative result without recreating the run or retrying the notification. Foreground calls never emit background work notifications because there is no detached work to report.
999
+
979
1000
  ## Related documentation
980
1001
 
981
1002
  - [Agent lifecycle](https://mastra.ai/docs/guides/agent-lifecycle): Full-run ordering and `RequestContext` visibility
@@ -143,9 +143,15 @@ To add your own endpoints, see [Custom API Routes](https://mastra.ai/docs/server
143
143
 
144
144
  ## Graceful shutdown and rolling deploys
145
145
 
146
- By default, the generated server handles `SIGINT` and `SIGTERM`. It stops accepting connections, waits up to [`server.drainTimeout`](https://mastra.ai/reference/configuration) for active requests and streams, then runs `mastra.shutdown()`. The drain timeout defaults to 5 seconds. A second signal terminates the process immediately. See [`server.handleShutdownSignals`](https://mastra.ai/reference/configuration) if you need to manage signals yourself.
146
+ By default, the generated server handles `SIGINT` and `SIGTERM`. It stops accepting connections, waits up to [`server.drainTimeout`](https://mastra.ai/reference/configuration) for active requests and streams, then runs `mastra.shutdown()`. Shutdown gives in-flight workflow runs (including durable agent runs) the same drain window to finish before workers and pub/sub subscriptions are torn down. The drain timeout defaults to 5 seconds. A second signal terminates the process immediately. See [`server.handleShutdownSignals`](https://mastra.ai/reference/configuration) if you need to manage signals yourself.
147
147
 
148
- Increase `drainTimeout` when your hosting platform's termination grace period can accommodate longer turns. Keep enough time after the drain for shutdown cleanup.
148
+ Increase `drainTimeout` when your hosting platform's termination grace period can accommodate longer turns. The HTTP drain and the workflow drain run one after the other, so keep enough time for both plus shutdown cleanup.
149
+
150
+ If you call `mastra.shutdown()` yourself, pass `drainTimeout` to control how long it waits for in-flight workflow runs:
151
+
152
+ ```typescript
153
+ await mastra.shutdown({ drainTimeout: 30_000 })
154
+ ```
149
155
 
150
156
  ```typescript
151
157
  import { Mastra } from '@mastra/core/mastra'
@@ -142,7 +142,11 @@ Deleting an item hides it from the current dataset version but retains its conte
142
142
  await dataset.purgeItem({ itemId: 'item-abc-123' })
143
143
  ```
144
144
 
145
- Purging replaces content in every historical row and deletion tombstone present during the operation with redacted values, and scrubs linked experiment-result payloads, tags, and comments. Later experiment-result submissions for the item are also stored with redacted content, and later `updateItem()` calls reject with `DATASET_ITEM_PURGED`. Don't run purge concurrently with dataset item updates or deletions because a write that started before purge can commit a stale revision afterward. Purging keeps version, identity, experiment counters, and review status so pinned dataset versions and experiment records remain structurally consistent. It doesn't create a dataset version and can't be undone. Avoid storing sensitive data in `externalId`, which remains unchanged as the item's identity key.
145
+ Purging replaces content in every historical row and deletion tombstone with redacted values, and scrubs linked experiment-result payloads, tags, and comments. Later experiment-result submissions for the item are also stored with redacted content.
146
+
147
+ Purge serializes or conflicts with concurrent dataset item writers without guaranteeing which operation completes first. If a mutating `updateItem()` call loses the race, it re-reads the purge marker and rejects with `DATASET_ITEM_PURGED`. `deleteItem()` remains idempotent, and any deletion tombstone created during the race stays redacted.
148
+
149
+ Normal item mutations use Slowly Changing Dimension Type 2 (SCD-2) versioning. Permanent purge intentionally overrides historical immutability for erasure while preserving item identity and the dataset version timeline. It doesn't create a dataset version and can't be undone. Experiment counters and review status are also preserved. Avoid storing sensitive data in `externalId`, which remains unchanged as the item's identity key.
146
150
 
147
151
  MongoDB storage requires a replica set or sharded deployment with transaction support for this operation. Purging fails before changing data when MongoDB transactions aren't available.
148
152
 
@@ -178,7 +178,7 @@ await assistant.generate('Help me plan the next project milestone.', {
178
178
 
179
179
  The example assumes storage is configured on the registered Mastra instance or directly on `Memory`. `lastMessages` controls how many recent messages Mastra loads from the thread. The default is 10.
180
180
 
181
- Message history works well for shorter conversations where recent turns contain the context the agent needs. For long-running conversations, Mastra recommends [Observational Memory](https://mastra.ai/docs/memory/observational-memory), which keeps recent conversation available and turns older history into a dense observation log.
181
+ Message history works well for shorter conversations where recent turns contain the context the agent needs. Because the `lastMessages` window slides forward on every request, each turn past the limit removes the oldest message from the start of the prompt and invalidates the provider prompt cache. For long-running conversations, Mastra recommends [Observational Memory](https://mastra.ai/docs/memory/observational-memory), which keeps recent conversation available and turns older history into a dense observation log while keeping the prompt prefix stable for caching.
182
182
 
183
183
  ## Observational Memory
184
184
 
@@ -44,16 +44,16 @@ The full set of options is listed in the [backgroundTasks configuration referenc
44
44
 
45
45
  ## Run a tool in the background
46
46
 
47
- Enabling the manager doesn't run anything in the background by itself as every tool defaults to foreground execution. Tools opt in at one of two layers:
47
+ Enabling the manager doesn't run anything in the background by itself. Tools become eligible at one of two layers:
48
48
 
49
49
  1. **Tool-level config**: the tool itself declares it as background-eligible.
50
50
  2. **Agent-level config**: the agent declares which of its tools are background-eligible.
51
51
 
52
- Once a tool has opted in, the LLM can optionally include a `_background` field in the tool arguments to override the resolved config for a specific call (timeout, retries, or to flip the call back to foreground).
52
+ Eligible tools default to `deferred` execution. Set `defaultDisposition: 'foreground'` at either layer when eligibility should only give the LLM the option to run a call in the background. The LLM can include a `_background` field in the tool arguments to select `foreground`, `deferred`, or `awaited` execution for a specific call and override its timeout or retries.
53
53
 
54
54
  ### Tool-level
55
55
 
56
- Set `background.enabled: true` on the tool definition. Tools opted in at this layer run in the background whenever called by an agent that has the manager enabled.
56
+ Set `background.enabled: true` on the tool definition. Tools opted in at this layer are eligible for background execution when called by an agent that has the manager enabled. Their configured default disposition determines whether each call runs inline or in the background unless the call includes a `_background` override.
57
57
 
58
58
  ```typescript
59
59
  import { createTool } from '@mastra/core/tools'
@@ -65,6 +65,7 @@ export const researchTool = createTool({
65
65
  inputSchema: z.object({ topic: z.string() }),
66
66
  background: {
67
67
  enabled: true,
68
+ defaultDisposition: 'deferred',
68
69
  timeoutMs: 600_000,
69
70
  maxRetries: 1,
70
71
  },
@@ -104,20 +105,23 @@ When a tool is registered on an agent that has background tasks enabled, the mod
104
105
  ```json
105
106
  {
106
107
  "topic": "solana",
107
- "_background": { "enabled": true, "timeoutMs": 900_000 }
108
+ "_background": { "disposition": "awaited", "timeoutMs": 900000 }
108
109
  }
109
110
  ```
110
111
 
111
- The `_background` override is a _modifier_ on tools the developer has already opted in at the tool or agent layer, it's not a standalone opt-in. If a tool hasn't been opted in, `_background.enabled: true` from the model is ignored and the tool runs in the foreground. This keeps deterministic, foreground-only tools (calculators, lookups, schema validators) from being silently dispatched as tasks.
112
+ The available dispositions are:
112
113
 
113
- ### Resolution order
114
+ - `foreground`: Run the tool synchronously without background-work lifecycle signals.
115
+ - `deferred`: Dispatch the tool and let the agent continue. The task remains attached to the run, and streams using `untilIdle` wait for it to reconcile.
116
+ - `awaited`: Dispatch the tool through the background task manager, but hold the current branch until its authoritative result has been reconciled.
117
+
118
+ For compatibility, `_background.enabled: true` selects `deferred`, and `_background.enabled: false` selects `foreground`. An explicit `disposition` takes precedence over `enabled`.
114
119
 
115
- When a tool call is dispatched, the resolved background config is computed in this priority order:
120
+ The `_background` override only _modifies_ tools the developer has already opted in at the tool or agent layer. If a tool hasn't been opted in, a model-selected background disposition is ignored and the tool runs in the foreground. This keeps deterministic, foreground-only tools (calculators, lookups, schema validators) from being silently dispatched as tasks.
116
121
 
117
- 1. Agent-level `backgroundTasks.tools` entry for the tool.
118
- 2. Tool-level `background` config.
119
- 3. LLM `_background.enabled` override (only used to enable background dispatch when the tool was opted in at one of the layers above).
120
- 4. Manager defaults (`defaultTimeoutMs`, `defaultRetries`).
122
+ ### Resolution order
123
+
124
+ When a tool call is dispatched, agent-level and tool-level settings determine eligibility and fallback values. For an eligible tool, the LLM `_background` fields override the corresponding values for that call. Manager defaults fill timeout and retry values that remain unset. An `_background` field can't enable a tool that's not eligible.
121
125
 
122
126
  If the agent has `backgroundTasks.disabled: true`, every tool call runs synchronously regardless of the layers above.
123
127
 
@@ -204,7 +208,7 @@ const stream = await supervisor.stream('Research AI in education and write an ar
204
208
 
205
209
  ### Inheriting from the subagent
206
210
 
207
- If a subagent isn't listed under the supervisor's `backgroundTasks.tools` but has background-eligible tools of its own (either via tool-level `background.enabled: true` or its own `backgroundTasks.tools` entry) the framework still dispatches the entire subagent invocation as a background task. The supervisor inherits the subagent's intent: the subagent itself becomes the background task, and its inner tools run in the foreground inside the subagent's loop.
211
+ If a subagent isn't listed under the supervisor's `backgroundTasks.tools` but has background-eligible tools of its own (either via tool-level `background.enabled: true` or its own `backgroundTasks.tools` entry) the framework still dispatches the entire subagent invocation as a background task. The supervisor inherits the subagent's intent: the subagent itself becomes the background task, and it can dispatch its eligible ordinary tools in the background inside its loop. Further delegated-agent calls run in the foreground.
208
212
 
209
213
  The background config used for the inherited dispatch (for example `waitTimeoutMs`) is derived from the subagent's own `backgroundTasks` config.
210
214
 
@@ -223,9 +227,11 @@ const researchAgent = new Agent({
223
227
  })
224
228
  ```
225
229
 
226
- When this `researchAgent` is delegated to from a supervisor that has no backgroundTask configuration for the `researchAgent`, the supervisor still dispatches the whole `researchAgent` invocation as a background task, and `deepResearchTool` runs in the foreground inside that invocation, instead of dispatching its own nested background task.
230
+ When this `researchAgent` is delegated to from a supervisor that has no background task configuration for the `researchAgent`, the supervisor still dispatches the whole `researchAgent` invocation as a background task.
231
+
232
+ Mastra supports one nested background-tool level inside a delegated run: a root agent can delegate to a subagent, and that subagent can dispatch an eligible ordinary tool in the background. A second delegated-agent edge runs in the foreground and doesn't receive nested background guidance. This limit is execution-scoped and doesn't mutate the shared subagent configuration.
227
233
 
228
- Use this pattern when you want a subagent to behave consistently in the background regardless of which supervisor invokes it. Use the supervisor-side opt-in (above) when you want to tune background behavior centrally per supervisor.
234
+ Which layer to use depends on where consistency matters: subagent-level configuration travels with the agent, so its background behavior stays the same under every supervisor, whereas the supervisor-side opt-in above centralizes that tuning in one place. This boundary stops at a single nested delegation level and has no workflow integration.
229
235
 
230
236
  ## Suspending and resuming
231
237
 
@@ -315,23 +321,23 @@ export const mastra = new Mastra({
315
321
  Calling `stream()` with no filter returns a stream of every task event in the system. On connection, the stream emits a snapshot of all currently running tasks, then forwards live events as they happen.
316
322
 
317
323
  ```typescript
318
- const bgManager = mastra.backgroundTaskManager
319
- if (!bgManager) throw new Error('Background tasks are not enabled')
324
+ const bgManager = mastra.backgroundTaskManager;
325
+ if (!bgManager) throw new Error('Background tasks are not enabled');
320
326
 
321
- const controller = new AbortController()
322
- const stream = bgManager.stream({ abortSignal: controller.signal })
327
+ const controller = new AbortController();
328
+ const stream = bgManager.stream({ abortSignal: controller.signal });
323
329
 
324
330
  for await (const chunk of stream) {
325
331
  switch (chunk.type) {
326
332
  case 'background-task-running':
327
- console.log('started', chunk.payload.taskId, chunk.payload.toolName)
328
- break
333
+ console.log('started', chunk.payload.taskId, chunk.payload.toolName);
334
+ break;
329
335
  case 'background-task-completed':
330
- console.log('done', chunk.payload.taskId, chunk.payload.result)
331
- break
336
+ console.log('done', chunk.payload.taskId, chunk.payload.result);
337
+ break;
332
338
  case 'background-task-failed':
333
- console.error('failed', chunk.payload.taskId, chunk.payload.error)
334
- break
339
+ console.error('failed', chunk.payload.taskId, chunk.payload.error);
340
+ break;
335
341
  }
336
342
  }
337
343
  ```
@@ -318,6 +318,45 @@ Use `createNotificationInboxTool()` to give agents one tool for inbox actions in
318
318
 
319
319
  `sendNotificationSignal()` requires a storage domain with `notifications` support. Use `sendSignal({ type: 'notification' })` only for lower-level notification-shaped context that should bypass inbox storage.
320
320
 
321
+ ## Cross-agent connections in Mastra Code
322
+
323
+ Cross-agent communication is experimental and off by default. Enable it in Mastra Code with the `/settings` toggle "Experimental cross-agent communication" and restart. Embedded clients can set `crossAgentSignals: true` when calling `createMastraCode()`. The setting enables thread ownership advertisements and peer discovery. It also makes the agent connection tools available. It doesn't affect the pub/sub transport itself.
324
+
325
+ ```typescript
326
+ import { createMastraCode } from 'mastracode'
327
+
328
+ const mastraCode = await createMastraCode({
329
+ crossAgentSignals: true,
330
+ })
331
+ ```
332
+
333
+ Mastra Code can use notification signals to communicate between agent threads. A peer is an exact `{ agentId, resourceId, threadId }` endpoint. Discovery is a point-in-time advertisement that the exact endpoint is available now. It isn't a durable connection or a guarantee that delivery will succeed.
334
+
335
+ `agent_connections_list` returns one `peers` collection. Each peer includes:
336
+
337
+ - `relationship`: `none` or `saved`. This is the authoritative durable state for the current sender thread.
338
+ - `presence`: `advertised` or `absent`. This reflects the current discovery pass.
339
+ - `displayStatus`: A derived, presentational value: `discovered`, `connected`, or `saved`.
340
+ - `canAttemptSend`: `true` only when the peer is both saved and currently advertised.
341
+
342
+ The displayed statuses mean:
343
+
344
+ | Status | Meaning | Can attempt a send? |
345
+ | -------------- | ----------------------------------------------------------- | ------------------- |
346
+ | `[discovered]` | Advertised now, but not saved by the current sender thread. | No |
347
+ | `[connected]` | Saved by the current sender thread and advertised now. | Yes |
348
+ | `[saved]` | Saved by the current sender thread, but not advertised now. | No |
349
+
350
+ Use the connection tools for distinct operations:
351
+
352
+ - `agent_connect` saves a freshly discovered exact peer in the current sender thread.
353
+ - `agent_disconnect` removes a saved peer from the current sender thread. The peer doesn't need to be currently advertised, and repeated disconnects are safe.
354
+ - `agent_signal_send` requires the exact peer to remain saved and freshly advertised at send time. A previously discovered or recently seen peer that's absent from the current discovery pass isn't sendable.
355
+
356
+ Saved connections remain in the sender thread until explicitly disconnected. Every send independently refreshes discovery and verifies the exact endpoint before routing begins. A saved connection therefore records collaboration intent. It doesn't indicate current presence.
357
+
358
+ After a send attempt reaches Core routing, the routing result is authoritative. The result can be `wake`, `deliver`, `persist`, `blocked`, or `discard`, depending on the target thread and notification policy. A send that isn't acknowledged by the advertised thread owner returns an error and doesn't consume its `messageId`, so the sender can retry it.
359
+
321
360
  ## Distributed and serverless deployments
322
361
 
323
362
  Signals coordinate runs through a pub/sub backend. When a signal arrives on a backend that implements `LeaseProvider`, Mastra acquires a lease on the target thread so a single process owns the conversation at a time, then either wakes the agent or routes the input into the running loop. Backends without leasing fall back to a no-op that always grants ownership, which is fine in a single process but not across instances.
@@ -74,7 +74,7 @@ Short list of known model IDs are:
74
74
 
75
75
  Go to <https://mastra.ai/models> for a full list of supported models.
76
76
 
77
- Add a tool an agent by importing the tool and passing it to the agent constructor as a tools object.
77
+ Add a tool to an agent by importing the tool and passing it to the agent constructor as a tools object.
78
78
 
79
79
  Example:
80
80
 
@@ -122,6 +122,8 @@ You can use this history in two ways:
122
122
  - **Automatic inclusion**: Mastra automatically includes recent messages in the context window. The default of 10 messages keeps agents grounded in the conversation. Adjust it with `lastMessages` when needed.
123
123
  - [**Manual querying**](#querying): For more control, query threads and messages directly with `recall()`. Use the results to choose which memories enter the context window or to render conversation history in your UI.
124
124
 
125
+ > **Note:** `lastMessages` counts every stored message, including tool calls, tool results, and [signals](https://mastra.ai/docs/harness/signals) of any kind, so a single turn can add several messages to the count. The window also slides forward on every request: once a thread grows past the limit, the oldest message leaves context on each turn, which changes the start of the prompt and invalidates the provider prompt cache. For long-running conversations, use [Observational Memory](https://mastra.ai/docs/memory/observational-memory), which keeps the prompt prefix stable.
126
+
125
127
  > **Tip:** When memory is enabled, [Studio](https://mastra.ai/docs/studio/overview) uses message history to display past conversations in the chat sidebar.
126
128
 
127
129
  ## Thread title generation
@@ -232,7 +234,7 @@ const thread = await memory.getThreadById({ threadId: 'thread-123' })
232
234
 
233
235
  Once you have a thread, use [`recall()`](https://mastra.ai/reference/memory/recall) to retrieve its messages. It supports pagination and [semantic search](https://mastra.ai/docs/memory/semantic-recall), with optional date filtering.
234
236
 
235
- Basic recall returns all messages from a thread:
237
+ Fetch a thread's history without pagination. Recall hides reminder signals by default; pass `hideSignals: false` to include them, `true` to hide all recognized signals, or an array to omit selected types. See [signal visibility and compatibility](https://mastra.ai/reference/memory/recall) for matching rules and precedence.
236
238
 
237
239
  ```typescript
238
240
  const { messages } = await memory.recall({
@@ -343,7 +345,9 @@ const { thread, clonedMessages } = await memory.cloneThread({
343
345
 
344
346
  You can filter cloned messages by count or date range and specify custom thread IDs. Utility methods are also available to inspect clone relationships.
345
347
 
346
- See [`cloneThread()`](https://mastra.ai/reference/memory/cloneThread) and [clone utilities](https://mastra.ai/reference/memory/clone-utilities) for the full API.
348
+ If you don't need the copied messages returned, for example when forking a long thread, use `copyThread()`. It never returns message payloads, and on LibSQL and PostgreSQL the rows are copied inside the database. When semantic recall is enabled, the copied messages are still read back in batches to generate embeddings.
349
+
350
+ See [`cloneThread()`](https://mastra.ai/reference/memory/cloneThread), [`copyThread()`](https://mastra.ai/reference/memory/copyThread), and [clone utilities](https://mastra.ai/reference/memory/clone-utilities) for the full API.
347
351
 
348
352
  ## Deleting messages
349
353
 
@@ -240,6 +240,31 @@ await parentAgent.generate('Research AI trends', {
240
240
  })
241
241
  ```
242
242
 
243
+ ### Reusing an earlier subagent result
244
+
245
+ By default, subagent results reach the parent agent as tool results, which are stripped from the context forwarded to later subagents. The parent agent must restate an earlier result in the next delegation prompt, which costs tokens and loses detail.
246
+
247
+ Set `enableResultReferences` to let a later delegation reuse an earlier result verbatim:
248
+
249
+ ```typescript
250
+ await parentAgent.generate('Find and fix the token refresh bug', {
251
+ delegation: {
252
+ enableResultReferences: true,
253
+ },
254
+ })
255
+ ```
256
+
257
+ When enabled:
258
+
259
+ - Each successful, non-empty subagent result gets a reference ID such as `explorer-1`. The parent agent's model sees it as a `[ref: explorer-1]` line after the subagent's text.
260
+ - The delegation tools gain a `contextFromRefs` input. The parent agent can pass earlier IDs, either as strings (`["explorer-1"]`) or as objects with an optional label and note (`[{ ref: "explorer-1", as: "investigation", note: "bug location" }]`).
261
+ - The referenced text is inserted before the delegation prompt, each result in its own labeled block, exactly as the earlier subagent produced it. `onDelegationStart` and `messageFilter` receive the expanded prompt.
262
+ - If `onDelegationComplete` returns `resultText`, the replaced text is what later delegations receive.
263
+
264
+ References are held in memory for a single parent agent run and aren't persisted. Rejected, failed, empty, and background-task delegations don't receive a reference ID. Unknown IDs are skipped with a warning and the delegation continues.
265
+
266
+ Referenced text is output from another agent. Each block uses a fresh, unpredictable tag and tells the receiving subagent to treat the contents as data, but if subagents handle untrusted input, add your own checks in `onDelegationStart` or through processors.
267
+
243
268
  ## Iteration monitoring
244
269
 
245
270
  `onIterationComplete` is called after each iteration of the parent agent's loop. Use it to monitor execution or guide the next iteration. You can also stop execution early.
@@ -1362,6 +1362,11 @@ export function NestedAgentChat() {
1362
1362
  <div key={index} className="nested-agent">
1363
1363
  <strong>Nested Agent: {id}</strong>
1364
1364
  {data.text && <p>{data.text}</p>}
1365
+ {data.toolErrors?.map(toolError => (
1366
+ <p key={toolError.toolCallId} className="nested-agent-error">
1367
+ {toolError.toolName} failed: {toolError.errorText}
1368
+ </p>
1369
+ ))}
1365
1370
  </div>
1366
1371
  )
1367
1372
  }
@@ -1388,6 +1393,8 @@ Key points:
1388
1393
  - Piping `fullStream` to `context.writer` creates `data-tool-agent` parts
1389
1394
  - Read `data-tool-agent-step` when you need the full payload for the nested step that finished
1390
1395
  - The `AgentDataPart` has `id` (on the part) and `data.text` (the current nested-agent text snapshot)
1396
+ - A tool that throws inside the nested agent is reported in `data.toolErrors`, each entry `{ toolCallId, toolName, args?, errorText, providerExecuted? }`. `errorText` is a JSON-safe string, so the failure survives serialization to the client instead of arriving as `{}`
1397
+ - `toolErrors` resets at each nested step boundary, like `toolCalls` and `toolResults`. For a finished step, read it from `data-tool-agent-step` (`data.step.toolErrors`) or from `data.steps[]` on the final snapshot
1391
1398
  - The tool still returns its own output after the stream completes
1392
1399
 
1393
1400
  For a complete implementation, see the [tool-nested-streams example](https://github.com/mastra-ai/ui-dojo/blob/main/src/pages/ai-sdk/tool-nested-streams.tsx) in UI Dojo.
@@ -111,7 +111,13 @@ const filesystem = new S3Filesystem({
111
111
  })
112
112
  ```
113
113
 
114
- Provider functions only apply to `S3Filesystem` API calls. When mounting the filesystem into an E2B sandbox, mount configuration only supports static `accessKeyId`, `secretAccessKey`, and `sessionToken` values, so credential refresh must be handled outside the mount.
114
+ Provider functions only apply to `S3Filesystem` API calls. When mounting the filesystem into an E2B or Daytona sandbox, mount configuration only supports static `accessKeyId`, `secretAccessKey`, and `sessionToken` values, so credential refresh must be handled outside the mount. See [temporary credentials in Daytona](https://mastra.ai/integrations/sandboxes/daytona) for mount lifetime and isolation requirements.
115
+
116
+ ### Prefix-scoped permissions
117
+
118
+ When `prefix` is set, initialization calls `ListObjectsV2` with the normalized prefix, including its trailing `/`, and `MaxKeys: 1`. Credentials must permit listing that prefix. Without a prefix, initialization uses `HeadBucket`. These checks verify access, not whether a directory exists.
119
+
120
+ The prefix limits which keys the filesystem addresses; it isn't an authorization boundary. For isolation, use credentials whose storage-provider policy restricts access to that prefix. Keep parent credentials on your backend. Read and write permissions alone aren't sufficient for prefixed filesystem initialization.
115
121
 
116
122
  ### Cloudflare R2
117
123
 
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Archil
6
6
 
7
- Stores files on [Archil](https://docs.archil.com) elastic, serverless disks. Combines an S3-compatible object API for fast reads/writes with `exec()` for POSIX shell operations and `grep()` for parallel server-side search. For interface details, see [WorkspaceFilesystem Interface](https://mastra.ai/reference/workspace/filesystem).
7
+ Stores files on [Archil](https://docs.archil.com) elastic, serverless disks. Combines an S3-compatible object API for fast reads/writes with `exec()` for POSIX shell operations and `diskGrep()` for parallel server-side search. For interface details, see [WorkspaceFilesystem Interface](https://mastra.ai/reference/workspace/filesystem).
8
8
 
9
9
  ## Installation
10
10
 
@@ -173,12 +173,12 @@ const result = await filesystem.exec('ls -la /data')
173
173
  // { exitCode: 0, stdout: '...', stderr: '' }
174
174
  ```
175
175
 
176
- #### `grep(options)`
176
+ #### `diskGrep(options)`
177
177
 
178
178
  Run a parallel server-side search across files on the disk.
179
179
 
180
180
  ```typescript
181
- const results = await filesystem.grep({
181
+ const results = await filesystem.diskGrep({
182
182
  directory: '/logs',
183
183
  pattern: 'ERROR',
184
184
  recursive: true,
@@ -430,7 +430,7 @@ export default function App(): React.JSX.Element {
430
430
  }
431
431
  ```
432
432
 
433
- This connects [`useChat()`](https://ai-sdk.dev/docs/reference/ai-sdk-ui/use-chat) to the `/chat/weather-agent` endpoint, sending propmts there and streaming the response back in chunks.
433
+ This connects [`useChat()`](https://ai-sdk.dev/docs/reference/ai-sdk-ui/use-chat) to the `/chat/weather-agent` endpoint, sending prompts there and streaming the response back in chunks.
434
434
 
435
435
  ## Test your agent
436
436
 
@@ -82,6 +82,8 @@ const result = await workspace.sandbox?.executeCommand?.('npm', ['test'], {
82
82
 
83
83
  Relative paths are resolved under `/workspace`. Absolute paths must also resolve within `/workspace`. Each file is sent as its own bridge request, and the bridge caps a single file at 32 MiB.
84
84
 
85
+ The Cloudflare sandbox cannot set per-file permissions, so a `writeFiles` call that includes a `mode` is rejected rather than silently ignored.
86
+
85
87
  ```typescript
86
88
  await workspace.sandbox?.writeFiles?.([
87
89
  { path: 'src/index.ts', content: "console.log('hello')\n" },
@@ -200,6 +200,39 @@ const workspace = new Workspace({
200
200
 
201
201
  When the workspace starts, the filesystems are automatically mounted at the specified paths. Code running in the sandbox can then access files at `/s3-data` and `/gcs-data` as if they were local directories.
202
202
 
203
+ #### Temporary S3 credentials
204
+
205
+ Pass all three credential values to `S3Filesystem` when using temporary credentials. Daytona forwards the session token to s3fs:
206
+
207
+ ```typescript
208
+ import { Workspace } from '@mastra/core/workspace'
209
+ import { DaytonaSandbox } from '@mastra/daytona'
210
+ import { S3Filesystem } from '@mastra/s3'
211
+
212
+ const workspace = new Workspace({
213
+ mounts: {
214
+ '/s3-data': new S3Filesystem({
215
+ bucket: process.env.S3_BUCKET!,
216
+ region: process.env.S3_REGION ?? 'us-east-1',
217
+ endpoint: process.env.S3_ENDPOINT,
218
+ prefix: 'resource-123/thread-456/',
219
+ accessKeyId: process.env.SCOPED_S3_ACCESS_KEY_ID!,
220
+ secretAccessKey: process.env.SCOPED_S3_SECRET_ACCESS_KEY!,
221
+ sessionToken: process.env.SCOPED_S3_SESSION_TOKEN!,
222
+ }),
223
+ },
224
+ sandbox: new DaytonaSandbox({ language: 'python', ephemeral: true }),
225
+ })
226
+ ```
227
+
228
+ Before starting the workspace, obtain credentials from your storage provider that authorize only the intended prefix. Credential issuance is provider-specific; Mastra doesn't mint or restrict credentials. A `prefix` selects a directory but doesn't enforce authorization. Authenticate each request, authorize its resource and thread, and use separate sandboxes for separate security scopes.
229
+
230
+ Temporary credentials are loaded when the mount starts and aren't automatically refreshed. Host-side credential provider functions don't refresh the mount. Keep runs within the credential lifetime and create a new sandbox with fresh credentials for later runs. Reconnecting to a sandbox doesn't renew its credentials.
231
+
232
+ Credentials are uploaded into owner-only files inside private directories. After launching s3fs, Daytona removes the temporary-credential staging file and directory. The daemon retains the credentials in its environment, which sandbox code running as the same user or root can still read. Never supply broader credentials than the sandbox needs. Mounts using long-lived credentials retain their password files while s3fs needs them. Unmount and reconnect cleanup remove those files after verifying that the daemon has exited. If a daemon is still active, including after a mount is moved aside, or its status can't be checked, the files are retained and a warning is logged. Cleanup on a later unmount of the same path retries removal; deleting the sandbox removes any remaining files.
233
+
234
+ Prefixed filesystems require permission to list their prefix during initialization. With s3fs, a prefixed mount may also require a zero-byte object at the exact `<prefix>/` key. Provision that directory marker before mounting. After launching s3fs, Daytona checks the mounted directory's metadata without listing its contents, with a 15-second timeout and forced termination after another 5 seconds. A failed check reports a mount failure. Cleanup attempts to unmount the failed mount, moving a stuck mount aside if necessary to free the original path for retry. Moved stale mounts may remain until sandbox deletion. This startup check doesn't guarantee read or write access to individual files or ongoing daemon health; verify a read and write through the mounted path.
235
+
203
236
  #### Via `sandbox.mount()`
204
237
 
205
238
  Mount manually at any point after the sandbox has started:
@@ -162,6 +162,19 @@ const sandbox = new DockerSandbox({
162
162
  })
163
163
  ```
164
164
 
165
+ ## Write files
166
+
167
+ Upload multiple files in one call with `writeFiles`. Relative paths resolve under the working directory. Set an optional per-file `mode` to control POSIX permissions; it must be an integer between `0o001` and `0o777`. When `mode` is omitted, new files are created with `0644`.
168
+
169
+ ```typescript
170
+ await sandbox.writeFiles([
171
+ { path: 'src/index.js', content: "console.log('hello')\n" },
172
+ { path: 'run.sh', content: '#!/bin/sh\n', mode: 0o755 },
173
+ ])
174
+ ```
175
+
176
+ Docker is the only built-in sandbox that applies an explicit `mode`. Providers that cannot honor per-file permissions (Vercel, E2B, Daytona, Cloudflare) reject a `writeFiles` call that includes a `mode` instead of silently dropping it.
177
+
165
178
  ## Bind mounts
166
179
 
167
180
  Mount host directories into the container using the `volumes` option:
@@ -205,6 +205,28 @@ export default createLiveKitWorker({
205
205
 
206
206
  `configuration.stt` works the same way for per-call transcription, for example a different transcription model or language per tenant. The greeting has a matching per-call form: `configuration.greeting.text` accepts a resolver with the same call context, so one worker can open with each tenant's own phrasing.
207
207
 
208
+ ### Per-call turn detection
209
+
210
+ LiveKit's `TurnDetector` classes read the job's inference executor when constructed, so they can only be created inside a LiveKit job, not at module scope where the worker options live. To use one, set the `configuration.turnDetection` resolver. It runs once per call with the same call context as `configuration.stt` and returns anything the top-level `turnDetection` option accepts. Return `undefined` to fall back to the top-level option.
211
+
212
+ ```typescript
213
+ import { turnDetector } from '@livekit/agents-plugin-livekit'
214
+
215
+ export default createLiveKitWorker({
216
+ mastra,
217
+ agent: 'support',
218
+ stt: 'deepgram/nova-3',
219
+ tts: 'cartesia/sonic-3',
220
+ turnDetection: 'multilingual',
221
+ configuration: {
222
+ // Constructed inside the job, where the inference executor is available.
223
+ turnDetection: () => new turnDetector.MultilingualModel(0.2),
224
+ },
225
+ })
226
+ ```
227
+
228
+ The semantic model's inference runners must be registered before the agent server boots, so when this resolver is set the worker imports `@livekit/agents-plugin-livekit` up front. Keep the top-level `turnDetection` set to `'multilingual'` or `'english'` to pre-register only that model; otherwise both stay available.
229
+
208
230
  ### Memory and threads
209
231
 
210
232
  When the resolved Mastra agent has memory configured, each call becomes one memory thread:
@@ -539,7 +561,7 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
539
561
 
540
562
  **vad** (`VAD | 'silero' | false`): Voice activity detection. 'silero' loads the Silero VAD from @livekit/agents-plugin-silero during prewarm. Pass an instance to bring your own, or false to disable. (Default: `'silero'`)
541
563
 
542
- **turnDetection** (`'multilingual' | 'english' | TurnDetectionMode`): End-of-turn detection. 'multilingual' and 'english' load LiveKit's semantic turn detector from @livekit/agents-plugin-livekit. Other values such as 'vad', 'stt', or 'manual' pass through.
564
+ **turnDetection** (`'multilingual' | 'english' | TurnDetectionMode`): End-of-turn detection. 'multilingual' and 'english' load LiveKit's semantic turn detector from @livekit/agents-plugin-livekit. Other values such as 'vad', 'stt', or 'manual' pass through. To construct a TurnDetector instance per call, set the configuration.turnDetection resolver — it takes precedence, with this option as the fallback.
543
565
 
544
566
  **turnHandling** (`Partial<TurnHandlingOptions>`): Turn handling tuning: endpointing delays, interruption sensitivity, preemptive generation. The worker disables preemptiveGeneration unless set here — each preemptive attempt re-runs the Mastra agent and persists partial user and assistant messages unless memory.options.readOnly is set.
545
567
 
@@ -551,7 +573,7 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
551
573
 
552
574
  **onTurnComplete** (`(ctx: VoiceTurnCompleteContext) => void | Promise<void>`): Called once per turn after the reply finished streaming to text-to-speech. Runs off the audio path and is not awaited. The context carries the produced reply (text, toolCalls, interrupted, usage) and the resolved memory mapping.
553
575
 
554
- **configuration** (`LiveKitWorkerConfiguration`): Grouped conversation and compliance configuration: the opening greeting and AI disclosure, consent requirements, agent-initiated hang-up, and per-call STT/TTS selection.
576
+ **configuration** (`LiveKitWorkerConfiguration`): Grouped conversation and compliance configuration: the opening greeting and AI disclosure, consent requirements, agent-initiated hang-up, and per-call STT/TTS/turn detection selection.
555
577
 
556
578
  **configuration.greeting** (`GreetingConfiguration`): The opening greeting and AI disclosure: text (a fixed string or a per-call resolver for per-tenant greetings), allowInterruptions, awaitPlayout, persist, and periodic re-disclosure via repeatEvery and repeatText.
557
579
 
@@ -563,6 +585,8 @@ if (process.argv[1] === fileURLToPath(import.meta.url)) {
563
585
 
564
586
  **configuration.tts** (`(context: VoiceCallContext) => TTS | string | undefined`): Per-call text-to-speech: a resolver invoked once per call (post-connect) with { metadata, requestContext, roomName, ctx }, returning anything the top-level tts option accepts — one voice or language per tenant. Return undefined to fall back to the top-level tts. Cache plugin instances across calls.
565
587
 
588
+ **configuration.turnDetection** (`(context: VoiceCallContext) => TurnDetectionMode | undefined`): Per-call end-of-turn detection: a resolver invoked once per call (post-connect, inside the LiveKit job) with { metadata, requestContext, roomName, ctx }, returning anything the top-level turnDetection option accepts. Use it to construct LiveKit TurnDetector instances, which need the job's inference executor. Return undefined to fall back to the top-level turnDetection.
589
+
566
590
  **greeting** (`string`): Static greeting spoken when the session starts. Deprecated: prefer configuration.greeting.text.
567
591
 
568
592
  **persistGreeting** (`boolean`): Save the spoken greeting to the memory thread as an assistant message, making the saved thread a faithful call transcript. Only applies when a greeting is set and memory is enabled. Deprecated: prefer configuration.greeting.persist. (Default: `true`)
@@ -4,7 +4,7 @@
4
4
 
5
5
  # ![Merge Gateway logo](https://models.dev/logos/merge-gateway.svg)Merge Gateway
6
6
 
7
- Merge Gateway aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 182 models through Mastra's model router.
7
+ Merge Gateway aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 187 models through Mastra's model router.
8
8
 
9
9
  Learn more in the [Merge Gateway documentation](https://docs.merge.dev/merge-gateway).
10
10
 
@@ -67,6 +67,7 @@ ANTHROPIC_API_KEY=ant-...
67
67
  | `deepseek/deepseek-v3.1` |
68
68
  | `deepseek/deepseek-v3.2` |
69
69
  | `deepseek/deepseek-v4-flash` |
70
+ | `deepseek/deepseek-v4-flash-0423` |
70
71
  | `deepseek/deepseek-v4-flash-0731` |
71
72
  | `deepseek/deepseek-v4-flash-0731-fast` |
72
73
  | `deepseek/deepseek-v4-pro` |
@@ -94,6 +95,9 @@ ANTHROPIC_API_KEY=ant-...
94
95
  | `google/gemini-embedding-001` |
95
96
  | `google/gemini-flash-latest` |
96
97
  | `google/gemini-flash-lite-latest` |
98
+ | `google/gemma-3-12b-it` |
99
+ | `google/gemma-3-27b-it` |
100
+ | `google/gemma-3-4b-it` |
97
101
  | `google/gemma-4-26b-a4b-it` |
98
102
  | `google/gemma-4-31b-it` |
99
103
  | `meta/llama-3.1-70b-instruct` |
@@ -160,6 +164,7 @@ ANTHROPIC_API_KEY=ant-...
160
164
  | `openai/gpt-oss-120b` |
161
165
  | `openai/gpt-oss-20b` |
162
166
  | `openai/gpt-oss-safeguard-120b` |
167
+ | `openai/gpt-oss-safeguard-20b` |
163
168
  | `openai/o1` |
164
169
  | `openai/o3` |
165
170
  | `openai/o3-mini` |
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Netlify
6
6
 
7
- Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access 252 models through Mastra's model router.
7
+ Netlify AI Gateway provides unified access to multiple providers with built-in caching and observability. Access 253 models through Mastra's model router.
8
8
 
9
9
  Learn more in the [Netlify documentation](https://docs.netlify.com/build/ai-gateway/overview/).
10
10
 
@@ -161,6 +161,7 @@ ANTHROPIC_API_KEY=ant-...
161
161
  | `openrouter/inclusionai/ling-3.0-flash-fin` |
162
162
  | `openrouter/inclusionai/ling-3.0-flash-fin:free` |
163
163
  | `openrouter/inclusionai/ling-3.0-flash-sante:free` |
164
+ | `openrouter/inclusionai/ling-3.0-flash-vl` |
164
165
  | `openrouter/inclusionai/ling-3.0-flash-vl:free` |
165
166
  | `openrouter/mancer/weaver` |
166
167
  | `openrouter/meta-llama/llama-3.1-70b-instruct` |