@mastra/mcp-docs-server 1.2.8-alpha.0 → 1.2.8-alpha.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.docs/docs/agent-builder/deploying.md +1 -0
- package/.docs/docs/agent-builder/integrations.md +1 -3
- package/.docs/docs/agent-builder/overview.md +1 -1
- package/.docs/docs/agent-controller/overview.md +1 -1
- package/.docs/docs/agents/overview.md +17 -13
- package/.docs/docs/agents/skills.md +1 -3
- package/.docs/docs/agents/structured-output.md +14 -7
- package/.docs/docs/agents/using-tools.md +80 -26
- package/.docs/docs/browser/agent-browser.md +1 -0
- package/.docs/docs/browser/firecrawl.md +129 -0
- package/.docs/docs/browser/overview.md +16 -2
- package/.docs/docs/capabilities/channels/other-adapters.md +1 -1
- package/.docs/docs/capabilities/channels/overview.md +1 -0
- package/.docs/docs/capabilities/channels/slack.md +113 -3
- package/.docs/docs/deployment/mastra-server.md +35 -0
- package/.docs/docs/deployment/monorepo.md +2 -0
- package/.docs/docs/deployment/overview.md +6 -0
- package/.docs/docs/deployment/sandbox.md +277 -0
- package/.docs/docs/editor/overview.md +2 -6
- package/.docs/docs/editor/tools.md +2 -6
- package/.docs/docs/evals/multi-turn.md +178 -0
- package/.docs/docs/evals/overview.md +1 -0
- package/.docs/docs/evals/running-in-ci.md +3 -9
- package/.docs/docs/getting-started/file-based-agents.md +2 -2
- package/.docs/docs/index.md +17 -13
- package/.docs/docs/long-running-agents/background-tasks.md +2 -2
- package/.docs/docs/long-running-agents/durable-agents.md +2 -1
- package/.docs/docs/long-running-agents/schedules.md +2 -2
- package/.docs/docs/mastra-platform/database.md +13 -6
- package/.docs/docs/mastra-platform/deploy.md +17 -8
- package/.docs/docs/mastra-platform/environments.md +5 -4
- package/.docs/docs/mastra-platform/overview.md +1 -1
- package/.docs/docs/mastra-platform/regions.md +73 -0
- package/.docs/docs/mastra-platform/workspace.md +111 -0
- package/.docs/docs/mcp/overview.md +1 -1
- package/.docs/docs/memory/multi-user-threads.md +4 -4
- package/.docs/docs/memory/observational-memory.md +78 -3
- package/.docs/docs/memory/overview.md +4 -0
- package/.docs/docs/memory/working-memory.md +1 -1
- package/.docs/docs/observability/integrations/bridges/otel.md +1 -3
- package/.docs/docs/observability/integrations/exporters/braintrust.md +34 -2
- package/.docs/docs/observability/integrations/exporters/otel.md +1 -3
- package/.docs/docs/server/auth/simple-auth.md +1 -3
- package/.docs/docs/server/request-context.md +2 -0
- package/.docs/docs/studio/overview.md +2 -0
- package/.docs/docs/voice/livekit.md +1 -1
- package/.docs/docs/workflows/control-flow.md +1 -3
- package/.docs/docs/workspace/filesystem.md +1 -0
- package/.docs/docs/workspace/sandbox.md +1 -0
- package/.docs/guides/build-your-ui/ai-sdk-ui.md +1 -0
- package/.docs/guides/guide/signal-provider.md +2 -0
- package/.docs/guides/migrations/vnext-to-standard-apis.md +1 -3
- package/.docs/models/environment-variables.md +1 -0
- package/.docs/models/gateways/openrouter.md +340 -345
- package/.docs/models/gateways/vercel.md +6 -5
- package/.docs/models/index.md +1 -1
- package/.docs/models/providers/alibaba-token-plan-cn.md +22 -21
- package/.docs/models/providers/alibaba-token-plan.md +22 -21
- package/.docs/models/providers/ambient.md +12 -10
- package/.docs/models/providers/baseten.md +3 -2
- package/.docs/models/providers/crossmodel.md +2 -1
- package/.docs/models/providers/deepinfra.md +4 -7
- package/.docs/models/providers/empiriolabs.md +2 -1
- package/.docs/models/providers/evroc.md +3 -2
- package/.docs/models/providers/google.md +2 -2
- package/.docs/models/providers/kilo.md +1 -1
- package/.docs/models/providers/kimi-for-coding.md +10 -11
- package/.docs/models/providers/llmgateway.md +3 -8
- package/.docs/models/providers/moonshotai-cn.md +3 -2
- package/.docs/models/providers/moonshotai.md +3 -2
- package/.docs/models/providers/nebius.md +3 -1
- package/.docs/models/providers/novita-ai.md +3 -1
- package/.docs/models/providers/ollama-cloud.md +25 -42
- package/.docs/models/providers/opencode-go.md +5 -3
- package/.docs/models/providers/orcarouter.md +1 -1
- package/.docs/models/providers/privatemode-ai.md +10 -10
- package/.docs/models/providers/thinkingmachines.md +73 -0
- package/.docs/models/providers/togetherai.md +2 -6
- package/.docs/models/providers/zenmux.md +3 -1
- package/.docs/models/providers.md +1 -0
- package/.docs/reference/acp/acp-agent.md +2 -0
- package/.docs/reference/acp/create-acp-tool.md +28 -0
- package/.docs/reference/agent-controller/agent-controller-class.md +1 -1
- package/.docs/reference/agent-controller/session.md +15 -0
- package/.docs/reference/agents/agent.md +14 -2
- package/.docs/reference/agents/durable-agent.md +23 -0
- package/.docs/reference/agents/generate.md +19 -1
- package/.docs/reference/browser/firecrawl-browser.md +71 -0
- package/.docs/reference/cli/mastra.md +14 -6
- package/.docs/reference/code-sdk/mount-agent-controller.md +1 -3
- package/.docs/reference/core/getMCPServer.md +1 -3
- package/.docs/reference/core/getMCPServerById.md +1 -3
- package/.docs/reference/core/getTool.md +33 -0
- package/.docs/reference/core/getToolById.md +48 -0
- package/.docs/reference/core/getWorkflow.md +1 -3
- package/.docs/reference/core/listMCPServers.md +2 -6
- package/.docs/reference/core/listTools.md +38 -0
- package/.docs/reference/core/mastra-class.md +1 -1
- package/.docs/reference/datasets/addItem.md +1 -3
- package/.docs/reference/datasets/addItems.md +1 -3
- package/.docs/reference/datasets/compareExperiments.md +1 -3
- package/.docs/reference/datasets/create.md +1 -3
- package/.docs/reference/datasets/dataset.md +2 -6
- package/.docs/reference/datasets/datasets-manager.md +5 -15
- package/.docs/reference/datasets/delete.md +1 -3
- package/.docs/reference/datasets/deleteExperiment.md +1 -3
- package/.docs/reference/datasets/deleteItem.md +1 -3
- package/.docs/reference/datasets/deleteItems.md +1 -3
- package/.docs/reference/datasets/get.md +1 -3
- package/.docs/reference/datasets/getDetails.md +1 -3
- package/.docs/reference/datasets/getExperiment.md +1 -3
- package/.docs/reference/datasets/getItem.md +1 -3
- package/.docs/reference/datasets/getItemHistory.md +1 -3
- package/.docs/reference/datasets/list.md +1 -3
- package/.docs/reference/datasets/listExperimentResults.md +1 -3
- package/.docs/reference/datasets/listExperiments.md +1 -3
- package/.docs/reference/datasets/listItems.md +1 -3
- package/.docs/reference/datasets/listVersions.md +1 -3
- package/.docs/reference/datasets/startExperiment.md +1 -3
- package/.docs/reference/datasets/startExperimentAsync.md +1 -3
- package/.docs/reference/datasets/update.md +1 -3
- package/.docs/reference/datasets/updateItem.md +1 -3
- package/.docs/reference/editor/mastra-editor.md +1 -3
- package/.docs/reference/evals/context-recall.md +205 -0
- package/.docs/reference/evals/create-scorer.md +4 -10
- package/.docs/reference/evals/mastra-scorer.md +1 -3
- package/.docs/reference/evals/run-evals.md +94 -2
- package/.docs/reference/file-based-agents/logger.md +1 -1
- package/.docs/reference/file-based-agents/tools.md +7 -6
- package/.docs/reference/index.md +8 -0
- package/.docs/reference/memory/observational-memory.md +21 -0
- package/.docs/reference/observability/tracing/processors/sensitive-data-filter.md +1 -3
- package/.docs/reference/processors/token-limiter-processor.md +1 -3
- package/.docs/reference/rag/database-config.md +12 -0
- package/.docs/reference/storage/clickhouse.md +1 -1
- package/.docs/reference/storage/convex.md +17 -10
- package/.docs/reference/storage/retention.md +1 -1
- package/.docs/reference/streaming/agents/stream.md +3 -1
- package/.docs/reference/tools/create-tool.md +78 -52
- package/.docs/reference/tools/mcp-client.md +104 -24
- package/.docs/reference/tools/mcp-server.md +143 -44
- package/.docs/reference/tools/task-tools.md +2 -2
- package/.docs/reference/tools/tavily.md +1 -1
- package/.docs/reference/tools/vector-query-tool.md +22 -0
- package/.docs/reference/vectors/mongodb.md +2 -0
- package/.docs/reference/vectors/turbopuffer.md +4 -0
- package/.docs/reference/voice/mistral.md +127 -0
- package/.docs/reference/workspace/daytona-sandbox.md +2 -6
- package/.docs/reference/workspace/platform-filesystem.md +182 -0
- package/.docs/reference/workspace/platform-sandbox.md +200 -0
- package/CHANGELOG.md +71 -0
- package/dist/{chunk-GLPCVXXO.js → chunk-HGADBLKG.js} +2 -2
- package/dist/{chunk-GLPCVXXO.js.map → chunk-HGADBLKG.js.map} +1 -1
- package/dist/index.js +1 -1
- package/dist/stdio.js +1 -1
- package/dist/tools/docs.d.ts.map +1 -1
- package/package.json +8 -8
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
> Discover all available pages from the documentation index: https://mastra.ai/llms.txt
|
|
2
|
+
|
|
3
|
+
# Context recall scorer
|
|
4
|
+
|
|
5
|
+
The `createContextRecallScorer()` function creates a scorer that evaluates how well retrieved context covers the claims in a ground-truth reference answer. It measures retrieval completeness by checking what fraction of the ground truth's claims are attributable to the retrieved context.
|
|
6
|
+
|
|
7
|
+
This scorer requires a ground-truth reference answer, making it suitable for labeled datasets in CI or test environments. When `groundTruth` is not provided in the run, the scorer returns a score of 0 rather than throwing an error.
|
|
8
|
+
|
|
9
|
+
## RAG retrieval evaluation
|
|
10
|
+
|
|
11
|
+
Ideal for evaluating retrieval completeness in RAG pipelines where:
|
|
12
|
+
|
|
13
|
+
- You need to verify the retriever fetches all necessary information
|
|
14
|
+
- You have labeled datasets with known-correct answers
|
|
15
|
+
- You want to catch regressions in what the retriever returns
|
|
16
|
+
|
|
17
|
+
## Dataset-driven testing
|
|
18
|
+
|
|
19
|
+
Use when running evaluations against curated test sets:
|
|
20
|
+
|
|
21
|
+
- CI pipelines with ground-truth labeled questions
|
|
22
|
+
- A/B testing retrieval strategies
|
|
23
|
+
- Benchmarking embedding models for coverage
|
|
24
|
+
|
|
25
|
+
## Parameters
|
|
26
|
+
|
|
27
|
+
**model** (`MastraModelConfig`): The language model to use for evaluating claim attribution
|
|
28
|
+
|
|
29
|
+
**options** (`ContextRecallMetricOptions`): Configuration options for the scorer
|
|
30
|
+
|
|
31
|
+
**Note**: Either `context` or `contextExtractor` must be provided. When both are provided, `contextExtractor` is used only if the run input and output are agent-shaped (`MastraDBMessage[]`); otherwise the scorer falls back to `context`.
|
|
32
|
+
|
|
33
|
+
## `.run()` returns
|
|
34
|
+
|
|
35
|
+
**score** (`number`): Recall score between 0 and scale (default 0-1), representing the fraction of ground-truth claims covered by the context
|
|
36
|
+
|
|
37
|
+
**reason** (`string`): Human-readable explanation of which ground-truth claims were and were not found in the context
|
|
38
|
+
|
|
39
|
+
## Scoring details
|
|
40
|
+
|
|
41
|
+
### Claim attribution
|
|
42
|
+
|
|
43
|
+
Context Recall uses a two-step LLM evaluation followed by a deterministic score calculation:
|
|
44
|
+
|
|
45
|
+
1. **Claim extraction**: The ground-truth answer is decomposed into atomic claims
|
|
46
|
+
2. **Attribution check**: Each claim is checked against the retrieval context for support
|
|
47
|
+
|
|
48
|
+
The score is then calculated as the ratio of attributed claims to total claims, multiplied by scale.
|
|
49
|
+
|
|
50
|
+
### Scoring formula
|
|
51
|
+
|
|
52
|
+
```text
|
|
53
|
+
Context Recall = attributed_claims / total_claims × scale
|
|
54
|
+
|
|
55
|
+
Where:
|
|
56
|
+
- attributed_claims = number of ground-truth claims supported by the context
|
|
57
|
+
- total_claims = total number of claims extracted from the ground truth
|
|
58
|
+
- Attribution is binary: a claim is either supported (yes) or not (no)
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
### Score interpretation
|
|
62
|
+
|
|
63
|
+
These ranges assume the default `scale` of 1. When using a custom scale, multiply accordingly.
|
|
64
|
+
|
|
65
|
+
- **0.9-1.0**: Excellent recall — context covers nearly all ground-truth claims
|
|
66
|
+
- **0.7-0.8**: Good recall — most claims are covered, minor gaps
|
|
67
|
+
- **0.4-0.6**: Moderate recall — significant information missing from context
|
|
68
|
+
- **0.1-0.3**: Poor recall — most ground-truth claims not found in context
|
|
69
|
+
- **0.0**: No recall — none of the ground-truth claims are in the context
|
|
70
|
+
|
|
71
|
+
### Reason analysis
|
|
72
|
+
|
|
73
|
+
The reason field explains:
|
|
74
|
+
|
|
75
|
+
- Which ground-truth claims were found in the context
|
|
76
|
+
- Which claims were missing and what information gaps exist
|
|
77
|
+
- Specific context pieces that supported attributed claims
|
|
78
|
+
|
|
79
|
+
### Optimization insights
|
|
80
|
+
|
|
81
|
+
Use results to:
|
|
82
|
+
|
|
83
|
+
- **Improve retrieval**: Identify what types of information the retriever misses
|
|
84
|
+
- **Tune chunk size**: Ensure chunks contain enough detail to cover ground-truth claims
|
|
85
|
+
- **Evaluate embeddings**: Test different embedding models for better information coverage
|
|
86
|
+
- **Expand knowledge base**: Add documents that cover frequently missed claims
|
|
87
|
+
|
|
88
|
+
### Example calculation
|
|
89
|
+
|
|
90
|
+
Given ground truth: "Einstein was born in 1879. He developed relativity. He won the Nobel Prize."
|
|
91
|
+
|
|
92
|
+
Claims extracted: 3
|
|
93
|
+
|
|
94
|
+
- "Einstein was born in 1879" → attributed (context mentions birthdate)
|
|
95
|
+
- "Einstein developed relativity" → attributed (context covers relativity)
|
|
96
|
+
- "Einstein won the Nobel Prize" → not attributed (context doesn't mention Nobel Prize)
|
|
97
|
+
|
|
98
|
+
Recall = 2/3 = 0.67
|
|
99
|
+
|
|
100
|
+
## Scorer configuration
|
|
101
|
+
|
|
102
|
+
### Dynamic context extraction
|
|
103
|
+
|
|
104
|
+
```typescript
|
|
105
|
+
const scorer = createContextRecallScorer({
|
|
106
|
+
model: 'openai/gpt-5.5',
|
|
107
|
+
options: {
|
|
108
|
+
contextExtractor: (input, output) => {
|
|
109
|
+
const query = input?.inputMessages?.[0]?.content || ''
|
|
110
|
+
const searchResults = vectorDB.search(query, { limit: 10 })
|
|
111
|
+
return searchResults.map(result => result.content)
|
|
112
|
+
},
|
|
113
|
+
scale: 1,
|
|
114
|
+
},
|
|
115
|
+
})
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### Static context evaluation
|
|
119
|
+
|
|
120
|
+
```typescript
|
|
121
|
+
const scorer = createContextRecallScorer({
|
|
122
|
+
model: 'openai/gpt-5.5',
|
|
123
|
+
options: {
|
|
124
|
+
context: [
|
|
125
|
+
'Document 1: Einstein was born on 14 March 1879 in Ulm, Germany.',
|
|
126
|
+
'Document 2: Einstein published the theory of special relativity in 1905.',
|
|
127
|
+
'Document 3: Einstein moved to the United States in 1933.',
|
|
128
|
+
],
|
|
129
|
+
},
|
|
130
|
+
})
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
## Example
|
|
134
|
+
|
|
135
|
+
Evaluate RAG retrieval completeness against a labeled dataset:
|
|
136
|
+
|
|
137
|
+
```typescript
|
|
138
|
+
import { runEvals } from '@mastra/core/evals'
|
|
139
|
+
import { createContextRecallScorer } from '@mastra/evals/scorers/prebuilt'
|
|
140
|
+
import { myAgent } from './agent'
|
|
141
|
+
|
|
142
|
+
const scorer = createContextRecallScorer({
|
|
143
|
+
model: 'openai/gpt-5.5',
|
|
144
|
+
options: {
|
|
145
|
+
contextExtractor: (input, output) => {
|
|
146
|
+
// Extract context from tool invocation results in the agent output
|
|
147
|
+
return output
|
|
148
|
+
.filter(msg => msg?.role === 'assistant')
|
|
149
|
+
.flatMap(msg => msg?.content?.toolInvocations ?? [])
|
|
150
|
+
.filter((tool: any) => tool.state === 'result')
|
|
151
|
+
.map((tool: any) => JSON.stringify(tool.result))
|
|
152
|
+
},
|
|
153
|
+
},
|
|
154
|
+
})
|
|
155
|
+
|
|
156
|
+
const result = await runEvals({
|
|
157
|
+
data: [
|
|
158
|
+
{
|
|
159
|
+
input: 'What are the health benefits of green tea?',
|
|
160
|
+
groundTruth:
|
|
161
|
+
'Green tea contains antioxidants that reduce inflammation, L-theanine that improves focus, and catechins that boost metabolism.',
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
input: 'How does photosynthesis work?',
|
|
165
|
+
groundTruth:
|
|
166
|
+
'Photosynthesis converts sunlight into chemical energy using chlorophyll in chloroplasts, producing glucose and oxygen from carbon dioxide and water.',
|
|
167
|
+
},
|
|
168
|
+
],
|
|
169
|
+
scorers: [scorer],
|
|
170
|
+
target: myAgent,
|
|
171
|
+
onItemComplete: ({ scorerResults }) => {
|
|
172
|
+
console.log({
|
|
173
|
+
score: scorerResults[scorer.id].score,
|
|
174
|
+
reason: scorerResults[scorer.id].reason,
|
|
175
|
+
})
|
|
176
|
+
},
|
|
177
|
+
})
|
|
178
|
+
|
|
179
|
+
console.log(result.scores)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
For more details on `runEvals`, see the [runEvals reference](https://mastra.ai/reference/evals/run-evals).
|
|
183
|
+
|
|
184
|
+
To add this scorer to an agent, see the [Scorers overview](https://mastra.ai/docs/evals/overview) guide.
|
|
185
|
+
|
|
186
|
+
## Comparison with context precision
|
|
187
|
+
|
|
188
|
+
Choose the right scorer for your needs:
|
|
189
|
+
|
|
190
|
+
| Use case | Context Recall | Context Precision |
|
|
191
|
+
| ------------------------- | ------------------------ | ----------------------------- |
|
|
192
|
+
| **What it measures** | Coverage of ground truth | Relevance of retrieved chunks |
|
|
193
|
+
| **Direction** | Ground truth → context | Context → ground truth |
|
|
194
|
+
| **Position sensitive** | No | Yes (rewards early placement) |
|
|
195
|
+
| **Requires ground truth** | Yes | Yes |
|
|
196
|
+
| **Failure mode caught** | Missing information | Irrelevant noise |
|
|
197
|
+
|
|
198
|
+
Use both together for a complete picture of retrieval quality: precision catches junk in the context, recall catches gaps.
|
|
199
|
+
|
|
200
|
+
## Related
|
|
201
|
+
|
|
202
|
+
- [Context Precision Scorer](https://mastra.ai/reference/evals/context-precision): Evaluates if retrieved context is relevant and well-ranked
|
|
203
|
+
- [Context Relevance Scorer](https://mastra.ai/reference/evals/context-relevance): Evaluates context usage and quality
|
|
204
|
+
- [Faithfulness Scorer](https://mastra.ai/reference/evals/faithfulness): Measures answer groundedness in context
|
|
205
|
+
- [Custom Scorers](https://mastra.ai/docs/evals/custom-scorers): Creating your own evaluation metrics
|
|
@@ -23,18 +23,12 @@ const scorer = createScorer({
|
|
|
23
23
|
instructions: 'You are an expert evaluator...',
|
|
24
24
|
},
|
|
25
25
|
})
|
|
26
|
-
.preprocess({
|
|
27
|
-
|
|
28
|
-
})
|
|
29
|
-
.analyze({
|
|
30
|
-
/* step config */
|
|
31
|
-
})
|
|
26
|
+
.preprocess({/* step config */})
|
|
27
|
+
.analyze({/* step config */})
|
|
32
28
|
.generateScore(({ run, results }) => {
|
|
33
29
|
// Return a number
|
|
34
30
|
})
|
|
35
|
-
.generateReason({
|
|
36
|
-
/* step config */
|
|
37
|
-
})
|
|
31
|
+
.generateReason({/* step config */})
|
|
38
32
|
```
|
|
39
33
|
|
|
40
34
|
## `createScorer` options
|
|
@@ -51,7 +45,7 @@ const scorer = createScorer({
|
|
|
51
45
|
|
|
52
46
|
**judge.instructions** (`string`): System prompt/instructions for the LLM.
|
|
53
47
|
|
|
54
|
-
**judge.jsonPromptInjection** (`boolean
|
|
48
|
+
**judge.jsonPromptInjection** (`boolean | 'system' | 'inline' | 'auto'`): Controls how the judge's structured output schema reaches the model. Defaults to 'auto', which uses native structured output when supported and inline prompt injection otherwise. Explicit values override automatic routing.
|
|
55
49
|
|
|
56
50
|
**judge.inputProcessors** (`Processor[]`): Input processors applied to the internal judge agent before its messages reach the model (e.g. redaction, validation).
|
|
57
51
|
|
|
@@ -33,9 +33,7 @@ const result = await scorer.run({
|
|
|
33
33
|
input: 'What is machine learning?',
|
|
34
34
|
output: 'Machine learning is a subset of artificial intelligence...',
|
|
35
35
|
runId: 'optional-run-id',
|
|
36
|
-
requestContext: {
|
|
37
|
-
/* optional context */
|
|
38
|
-
},
|
|
36
|
+
requestContext: {/* optional context */},
|
|
39
37
|
})
|
|
40
38
|
```
|
|
41
39
|
|
|
@@ -31,6 +31,28 @@ console.log(`Average scores:`, result.scores)
|
|
|
31
31
|
console.log(`Processed ${result.summary.totalItems} items`)
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
+
### Multi-turn evaluation
|
|
35
|
+
|
|
36
|
+
```typescript
|
|
37
|
+
import { runEvals } from '@mastra/core/evals'
|
|
38
|
+
import { checks } from '@mastra/evals/checks'
|
|
39
|
+
import { weatherAgent } from './agents/weather-agent'
|
|
40
|
+
|
|
41
|
+
const result = await runEvals({
|
|
42
|
+
target: weatherAgent,
|
|
43
|
+
data: [
|
|
44
|
+
{
|
|
45
|
+
inputs: [
|
|
46
|
+
'What is the weather in Brooklyn?',
|
|
47
|
+
'What about tomorrow?',
|
|
48
|
+
'Compare the two forecasts.',
|
|
49
|
+
],
|
|
50
|
+
},
|
|
51
|
+
],
|
|
52
|
+
scorers: [checks.calledTool('get_weather', { times: 2 }), checks.includes('Brooklyn')],
|
|
53
|
+
})
|
|
54
|
+
```
|
|
55
|
+
|
|
34
56
|
### With gates and thresholds
|
|
35
57
|
|
|
36
58
|
```typescript
|
|
@@ -60,7 +82,7 @@ result.thresholdResults // [{ id, passed, averageScore, threshold }]
|
|
|
60
82
|
|
|
61
83
|
**gates** (`MastraScorer[]`): Scorers that must score 1.0 for the run to pass. If any gate averages below 1.0 across data items, the verdict is failed. Gates run before regular scorers on each data item. When provided, scorers may be omitted.
|
|
62
84
|
|
|
63
|
-
**targetOptions** (`AgentExecutionOptions | WorkflowRunOptions`): Options forwarded to the target during execution. For agents: options passed to agent.generate() (e.g. maxSteps, modelSettings, instructions). For workflows: options passed to run.start() (e.g. perStep, outputOptions, initialState).
|
|
85
|
+
**targetOptions** (`AgentExecutionOptions | WorkflowRunOptions`): Options forwarded to the target during execution. For agents: options passed to agent.generate() (e.g. maxSteps, modelSettings, instructions). For workflows: options passed to run.start() (e.g. perStep, outputOptions, initialState). For multi-turn agent runs (inputs/turns), runEvals generates and injects the shared thread and a resource, so memory.thread is optional; provide memory.resource to reuse a specific resource.
|
|
64
86
|
|
|
65
87
|
**concurrency** (`number`): Number of test cases to run concurrently. (Default: `1`)
|
|
66
88
|
|
|
@@ -68,7 +90,11 @@ result.thresholdResults // [{ id, passed, averageScore, threshold }]
|
|
|
68
90
|
|
|
69
91
|
## Data item structure
|
|
70
92
|
|
|
71
|
-
**input** (`string | string[] | CoreMessage[] | any`): Input data for the target. For agents: messages or strings. For workflows: workflow input data.
|
|
93
|
+
**input** (`string | string[] | CoreMessage[] | any`): Input data for the target. For agents: messages or strings. For workflows: workflow input data. Optional when inputs is provided.
|
|
94
|
+
|
|
95
|
+
**inputs** (`(string | string[] | CoreMessage[] | any)[]`): Multi-turn inputs. Each entry is one turn (same shape as input) sent sequentially to the agent on the same thread. Scorers see the accumulated output from all turns. Only supported for Agent targets. When provided, input can be omitted. Mutually exclusive with turns.
|
|
96
|
+
|
|
97
|
+
**turns** (`EvalTurn[]`): Multi-turn conversation with per-turn assertions. Each turn is an object { input, gates?, scorers? } sent sequentially on the same thread; its gates/scorers evaluate only that turn's input and output. Per-turn outcomes are reported in turnResults and folded into the overall verdict. Only supported for Agent targets. Mutually exclusive with input and inputs.
|
|
72
98
|
|
|
73
99
|
**groundTruth** (`any`): Expected or reference output for comparison during scoring.
|
|
74
100
|
|
|
@@ -112,6 +138,18 @@ For workflows, use `WorkflowScorerConfig` to specify scorers at different levels
|
|
|
112
138
|
|
|
113
139
|
**thresholdResults** (`ThresholdResult[]`): Per-threshold-scorer results averaged across all data items. Each entry has id, passed, averageScore, and threshold.
|
|
114
140
|
|
|
141
|
+
**turnResults** (`TurnResult[]`): Present when any data item uses turns. Each entry has index (zero-based turn), optional gateResults, thresholdResults, and scores (bare-scorer averages keyed by scorer id), aggregated by turn index across data items.
|
|
142
|
+
|
|
143
|
+
## EvalTurn
|
|
144
|
+
|
|
145
|
+
A single turn in a `turns` array. Its `gates`/`scorers` evaluate only that turn's input and output:
|
|
146
|
+
|
|
147
|
+
**input** (`string | string[] | CoreMessage[] | any`): The input sent to the agent for this turn.
|
|
148
|
+
|
|
149
|
+
**gates** (`MastraScorer[]`): Gates that must score 1.0 for this turn. A failing turn gate makes the overall verdict failed.
|
|
150
|
+
|
|
151
|
+
**scorers** (`ScorerEntry[]`): Scorers (optionally with thresholds) evaluated against this turn only. A missed per-turn threshold (with gates passing) makes the verdict scored.
|
|
152
|
+
|
|
115
153
|
## ScorerEntry
|
|
116
154
|
|
|
117
155
|
A scorer entry in the `scorers` array can be either a bare scorer or a scorer with a threshold:
|
|
@@ -314,8 +352,62 @@ const result = await runEvals({
|
|
|
314
352
|
})
|
|
315
353
|
```
|
|
316
354
|
|
|
355
|
+
### Multi-turn conversation evaluation
|
|
356
|
+
|
|
357
|
+
Use `inputs` to send sequential turns on a shared thread. Scorers see the accumulated output from all turns:
|
|
358
|
+
|
|
359
|
+
```typescript
|
|
360
|
+
const result = await runEvals({
|
|
361
|
+
target: chatAgent,
|
|
362
|
+
data: [
|
|
363
|
+
{
|
|
364
|
+
inputs: ['My favorite city is Brooklyn.', 'What is the weather in my favorite city?'],
|
|
365
|
+
},
|
|
366
|
+
],
|
|
367
|
+
gates: [checks.calledTool('get_weather')],
|
|
368
|
+
scorers: [{ scorer: checks.similarity('Brooklyn weather forecast'), threshold: 0.5 }],
|
|
369
|
+
})
|
|
370
|
+
|
|
371
|
+
// result.verdict: 'passed' | 'scored' | 'failed'
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
Each turn runs `agent.generate()` with the same `threadId`, so the agent sees the full conversation history. `runEvals` also injects a `resourceId` (Mastra memory scopes messages by resource + thread), defaulting it to the generated thread; pass `targetOptions.memory.resource` to pin a specific one. Cross-turn recall requires the agent to have a memory store configured; otherwise turns run in isolation. Mix single-turn (`input`) and multi-turn (`inputs`) items in the same `data` array. When using `inputs`, `input` can be omitted.
|
|
375
|
+
|
|
376
|
+
Scoring uses the accumulated output from all turns as `run.output`, but only the first turn as `run.input`. Prefer output-based scorers (`checks.includes`, `checks.calledTool`, `checks.similarity`) for multi-turn; input-relative scorers (e.g. faithfulness) only see the first turn's input. Trajectory scorers that read from the trace (`AgentScorerConfig.trajectory`) resolve against the last turn's span; tool-call checks that read `run.output` (like `checks.calledTool`) still see every turn.
|
|
377
|
+
|
|
378
|
+
### Per-turn assertions
|
|
379
|
+
|
|
380
|
+
Use `turns` to attach `gates`/`scorers` to individual turns. Each per-turn assertion sees only that turn's input and output, so a later-turn regression can't be hidden by an earlier turn:
|
|
381
|
+
|
|
382
|
+
```typescript
|
|
383
|
+
const result = await runEvals({
|
|
384
|
+
target: chatAgent,
|
|
385
|
+
data: [
|
|
386
|
+
{
|
|
387
|
+
turns: [
|
|
388
|
+
{
|
|
389
|
+
input: 'What is the weather in Brooklyn?',
|
|
390
|
+
gates: [checks.calledTool('get_weather')],
|
|
391
|
+
},
|
|
392
|
+
{
|
|
393
|
+
input: 'What about tomorrow?',
|
|
394
|
+
gates: [checks.calledTool('get_weather')], // must call again this turn
|
|
395
|
+
scorers: [{ scorer: checks.similarity('tomorrow forecast'), threshold: 0.5 }],
|
|
396
|
+
},
|
|
397
|
+
],
|
|
398
|
+
},
|
|
399
|
+
],
|
|
400
|
+
})
|
|
401
|
+
|
|
402
|
+
result.verdict // folds in per-turn gate/threshold outcomes
|
|
403
|
+
result.turnResults // [{ index, gateResults, thresholdResults, scores }]
|
|
404
|
+
```
|
|
405
|
+
|
|
406
|
+
Per-turn gates/scorers evaluate only that turn (`run.input`/`run.output` are that turn's). A failing turn gate makes the verdict `failed`; a missed turn threshold (gates passing) makes it `scored`. Top-level `scorers`/`gates` still score the accumulated conversation holistically. `turns` is Agent-only and cannot be combined with `input` or `inputs`.
|
|
407
|
+
|
|
317
408
|
## Related
|
|
318
409
|
|
|
410
|
+
- [Multi-turn evals](https://mastra.ai/docs/evals/multi-turn): Conceptual guide to multi-turn evaluation
|
|
319
411
|
- [Gates and Verdicts](https://mastra.ai/docs/evals/gates-and-verdicts): Conceptual guide to severity semantics
|
|
320
412
|
- [Quick Checks](https://mastra.ai/reference/evals/checks): Zero-LLM composable micro-scorers
|
|
321
413
|
- [createScorer()](https://mastra.ai/reference/evals/create-scorer): Create custom scorers for experiments
|
|
@@ -19,7 +19,7 @@ export default new PinoLogger({
|
|
|
19
19
|
})
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
Mastra registers the logger before storage, observability, and file-based agents, so those primitives log through this logger as they
|
|
22
|
+
Mastra registers the logger before storage, observability, and file-based agents, so those primitives log through this logger as they're wired up.
|
|
23
23
|
|
|
24
24
|
## Precedence with code
|
|
25
25
|
|
|
@@ -16,16 +16,17 @@ import { z } from 'zod'
|
|
|
16
16
|
|
|
17
17
|
export default createTool({
|
|
18
18
|
id: 'get_weather',
|
|
19
|
-
description: 'Get the current weather for a
|
|
19
|
+
description: 'Get the current weather for a location.',
|
|
20
20
|
inputSchema: z.object({
|
|
21
|
-
|
|
21
|
+
location: z.string(),
|
|
22
22
|
}),
|
|
23
23
|
outputSchema: z.object({
|
|
24
|
-
|
|
25
|
-
|
|
24
|
+
location: z.string(),
|
|
25
|
+
temperatureCelsius: z.number(),
|
|
26
|
+
conditions: z.string(),
|
|
26
27
|
}),
|
|
27
|
-
execute: async ({
|
|
28
|
-
return {
|
|
28
|
+
execute: async ({ location }) => {
|
|
29
|
+
return { location, temperatureCelsius: 21, conditions: 'sunny' }
|
|
29
30
|
},
|
|
30
31
|
})
|
|
31
32
|
```
|
package/.docs/reference/index.md
CHANGED
|
@@ -57,6 +57,7 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
57
57
|
- [WorkOS](https://mastra.ai/reference/auth/workos)
|
|
58
58
|
- [AgentBrowser](https://mastra.ai/reference/browser/agent-browser)
|
|
59
59
|
- [BrowserViewer](https://mastra.ai/reference/browser/browser-viewer)
|
|
60
|
+
- [FirecrawlBrowser](https://mastra.ai/reference/browser/firecrawl-browser)
|
|
60
61
|
- [MastraBrowser Class](https://mastra.ai/reference/browser/mastra-browser)
|
|
61
62
|
- [StagehandBrowser](https://mastra.ai/reference/browser/stagehand-browser)
|
|
62
63
|
- [ChannelProvider](https://mastra.ai/reference/channels/channel-provider)
|
|
@@ -97,6 +98,8 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
97
98
|
- [.getServer()](https://mastra.ai/reference/core/getServer)
|
|
98
99
|
- [.getStorage()](https://mastra.ai/reference/core/getStorage)
|
|
99
100
|
- [.getTelemetry()](https://mastra.ai/reference/core/getTelemetry)
|
|
101
|
+
- [.getTool()](https://mastra.ai/reference/core/getTool)
|
|
102
|
+
- [.getToolById()](https://mastra.ai/reference/core/getToolById)
|
|
100
103
|
- [.getVector()](https://mastra.ai/reference/core/getVector)
|
|
101
104
|
- [.getWorkflow()](https://mastra.ai/reference/core/getWorkflow)
|
|
102
105
|
- [.listAgents()](https://mastra.ai/reference/core/listAgents)
|
|
@@ -106,6 +109,7 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
106
109
|
- [.listMCPServers()](https://mastra.ai/reference/core/listMCPServers)
|
|
107
110
|
- [.listMemory()](https://mastra.ai/reference/core/listMemory)
|
|
108
111
|
- [.listScorers()](https://mastra.ai/reference/core/listScorers)
|
|
112
|
+
- [.listTools()](https://mastra.ai/reference/core/listTools)
|
|
109
113
|
- [.listVectors()](https://mastra.ai/reference/core/listVectors)
|
|
110
114
|
- [.listWorkflows()](https://mastra.ai/reference/core/listWorkflows)
|
|
111
115
|
- [.removeWorkspace()](https://mastra.ai/reference/core/removeWorkspace)
|
|
@@ -139,6 +143,7 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
139
143
|
- [Completeness](https://mastra.ai/reference/evals/completeness)
|
|
140
144
|
- [Content Similarity Scorer](https://mastra.ai/reference/evals/content-similarity)
|
|
141
145
|
- [Context Precision Scorer](https://mastra.ai/reference/evals/context-precision)
|
|
146
|
+
- [Context Recall Scorer](https://mastra.ai/reference/evals/context-recall)
|
|
142
147
|
- [Context Relevance Scorer](https://mastra.ai/reference/evals/context-relevance)
|
|
143
148
|
- [Faithfulness](https://mastra.ai/reference/evals/faithfulness)
|
|
144
149
|
- [Hallucination](https://mastra.ai/reference/evals/hallucination)
|
|
@@ -333,6 +338,7 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
333
338
|
- [Inworld Realtime](https://mastra.ai/reference/voice/inworld-realtime)
|
|
334
339
|
- [LiveKit](https://mastra.ai/reference/voice/livekit)
|
|
335
340
|
- [Mastra Voice](https://mastra.ai/reference/voice/mastra-voice)
|
|
341
|
+
- [Mistral](https://mastra.ai/reference/voice/mistral)
|
|
336
342
|
- [Murf](https://mastra.ai/reference/voice/murf)
|
|
337
343
|
- [OpenAI](https://mastra.ai/reference/voice/openai)
|
|
338
344
|
- [OpenAI Realtime](https://mastra.ai/reference/voice/openai-realtime)
|
|
@@ -389,6 +395,8 @@ The Reference section provides documentation of Mastra's API, including paramete
|
|
|
389
395
|
- [LocalSandbox](https://mastra.ai/reference/workspace/local-sandbox)
|
|
390
396
|
- [MesaFilesystem](https://mastra.ai/reference/workspace/mesa-filesystem)
|
|
391
397
|
- [ModalSandbox](https://mastra.ai/reference/workspace/modal-sandbox)
|
|
398
|
+
- [PlatformFilesystem](https://mastra.ai/reference/workspace/platform-filesystem)
|
|
399
|
+
- [PlatformSandbox](https://mastra.ai/reference/workspace/platform-sandbox)
|
|
392
400
|
- [RailwaySandbox](https://mastra.ai/reference/workspace/railway-sandbox)
|
|
393
401
|
- [S3Filesystem](https://mastra.ai/reference/workspace/s3-filesystem)
|
|
394
402
|
- [SandboxProcessManager](https://mastra.ai/reference/workspace/process-manager)
|
|
@@ -408,6 +408,27 @@ Setting `bufferTokens: false` disables both observation and reflection async buf
|
|
|
408
408
|
|
|
409
409
|
Observational Memory emits typed data parts during agent execution that clients can use for real-time UI feedback. These are streamed alongside the agent's response.
|
|
410
410
|
|
|
411
|
+
### Read extractor results
|
|
412
|
+
|
|
413
|
+
Both completion events carry extractor output in their `data` payload. The extractor fields are:
|
|
414
|
+
|
|
415
|
+
```typescript
|
|
416
|
+
interface DataOmObservationEndPart {
|
|
417
|
+
type: 'data-om-observation-end'
|
|
418
|
+
data: {
|
|
419
|
+
/** Whether the completed work was an observation or reflection */
|
|
420
|
+
operationType: 'observation' | 'reflection'
|
|
421
|
+
/** Values extracted during this OM operation, keyed by extractor slug */
|
|
422
|
+
extractedValues?: Record<string, unknown>
|
|
423
|
+
/** Extractor failures from this OM operation. Successful extractor values are still included */
|
|
424
|
+
extractionFailures?: Array<{ slug: string; error: string }>
|
|
425
|
+
// ...other fields documented in the tables below
|
|
426
|
+
}
|
|
427
|
+
}
|
|
428
|
+
```
|
|
429
|
+
|
|
430
|
+
Both extractor fields are optional. A completion can include values, failures, both, or neither. `data-om-observation-end` reports synchronous completion. `data-om-buffering-end` reports completed background work whose buffered content still awaits activation, although extractor metadata is already persisted. `DataOmBufferingEndPart` carries the same extractor fields, and both types are exported from `@mastra/memory/processors`. See [Read extracted values from a stream](https://mastra.ai/docs/memory/observational-memory) for a consumer example.
|
|
431
|
+
|
|
411
432
|
### `data-om-status`
|
|
412
433
|
|
|
413
434
|
Emitted once per agent loop step, before model generation. Provides a snapshot of the current memory state, including token usage for both context windows and the state of any async buffered content.
|
|
@@ -14,9 +14,7 @@ To opt out or customize the auto-applied filter, use the `sensitiveDataFilter` o
|
|
|
14
14
|
import { Observability } from '@mastra/observability'
|
|
15
15
|
|
|
16
16
|
new Observability({
|
|
17
|
-
configs: {
|
|
18
|
-
/* ... */
|
|
19
|
-
},
|
|
17
|
+
configs: {/* ... */},
|
|
20
18
|
// disable the auto-applied filter
|
|
21
19
|
sensitiveDataFilter: false,
|
|
22
20
|
// or customize it
|
|
@@ -85,9 +85,7 @@ export const agent = new Agent({
|
|
|
85
85
|
name: 'context-limited-agent',
|
|
86
86
|
instructions: 'You are a helpful assistant',
|
|
87
87
|
model: 'openai/gpt-5.5',
|
|
88
|
-
memory: new Memory({
|
|
89
|
-
/* ... */
|
|
90
|
-
}),
|
|
88
|
+
memory: new Memory({/* ... */}),
|
|
91
89
|
inputProcessors: [
|
|
92
90
|
new TokenLimiterProcessor({ limit: 4000 }), // Limits historical messages to ~4000 tokens
|
|
93
91
|
],
|
|
@@ -11,6 +11,7 @@ export type DatabaseConfig = {
|
|
|
11
11
|
pinecone?: PineconeConfig
|
|
12
12
|
pgvector?: PgVectorConfig
|
|
13
13
|
chroma?: ChromaConfig
|
|
14
|
+
turbopuffer?: TurbopufferConfig
|
|
14
15
|
[key: string]: any // Extensible for future databases
|
|
15
16
|
}
|
|
16
17
|
```
|
|
@@ -90,6 +91,17 @@ whereDocument: { "$contains": "API documentation" }
|
|
|
90
91
|
- Content-based document filtering
|
|
91
92
|
- Complex query combinations
|
|
92
93
|
|
|
94
|
+
### `TurbopufferConfig`
|
|
95
|
+
|
|
96
|
+
Configuration options specific to Turbopuffer vector store.
|
|
97
|
+
|
|
98
|
+
**consistency** (`'strong' | 'eventual'`): The consistency level for queries. "strong" (default) guarantees queries see all data written before the query started, at the cost of higher latency. "eventual" offers lower latency, but recently written data may not be visible yet.
|
|
99
|
+
|
|
100
|
+
**Use Cases:**
|
|
101
|
+
|
|
102
|
+
- Latency-sensitive queries where slightly stale data is acceptable (`eventual`)
|
|
103
|
+
- Read-your-writes workflows that must see the latest data (`strong`)
|
|
104
|
+
|
|
93
105
|
## Usage examples
|
|
94
106
|
|
|
95
107
|
**Basic Usage**:
|
|
@@ -131,7 +131,7 @@ bun x mastra migrate
|
|
|
131
131
|
|
|
132
132
|
The migration copies span data from `mastra_ai_spans` into `mastra_span_events` in day-sized batches. It handles column mapping, deduplicates legacy rows, and preserves the original table as a backup. After migration, traces appear in Studio through the vNext adapter.
|
|
133
133
|
|
|
134
|
-
> **Note:** The legacy table
|
|
134
|
+
> **Note:** The legacy table isn't deleted. Drop it manually after verifying the migration.
|
|
135
135
|
|
|
136
136
|
### ClickHouse for every domain
|
|
137
137
|
|
|
@@ -50,6 +50,7 @@ import {
|
|
|
50
50
|
mastraResourcesTable,
|
|
51
51
|
mastraWorkflowSnapshotsTable,
|
|
52
52
|
mastraScoresTable,
|
|
53
|
+
mastraObservationalMemoryTable,
|
|
53
54
|
mastraVectorIndexesTable,
|
|
54
55
|
mastraVectorsTable,
|
|
55
56
|
mastraCacheTable,
|
|
@@ -63,6 +64,7 @@ export default defineSchema({
|
|
|
63
64
|
mastra_resources: mastraResourcesTable,
|
|
64
65
|
mastra_workflow_snapshots: mastraWorkflowSnapshotsTable,
|
|
65
66
|
mastra_scorers: mastraScoresTable,
|
|
67
|
+
mastra_observational_memory: mastraObservationalMemoryTable,
|
|
66
68
|
mastra_vector_indexes: mastraVectorIndexesTable,
|
|
67
69
|
mastra_vectors: mastraVectorsTable,
|
|
68
70
|
mastra_cache: mastraCacheTable,
|
|
@@ -192,16 +194,21 @@ Use a non-empty `keyPrefix` unless you intentionally want `clear()` to remove ev
|
|
|
192
194
|
|
|
193
195
|
The storage implementation uses typed Convex tables for each Mastra domain:
|
|
194
196
|
|
|
195
|
-
| Domain
|
|
196
|
-
|
|
|
197
|
-
| Threads
|
|
198
|
-
| Messages
|
|
199
|
-
| Resources
|
|
200
|
-
|
|
|
201
|
-
|
|
|
202
|
-
|
|
|
203
|
-
| Cache
|
|
204
|
-
|
|
|
197
|
+
| Domain | Convex Table | Purpose |
|
|
198
|
+
| -------------------- | ----------------------------- | ----------------------------------------- |
|
|
199
|
+
| Threads | `mastra_threads` | Conversation threads |
|
|
200
|
+
| Messages | `mastra_messages` | Chat messages |
|
|
201
|
+
| Resources | `mastra_resources` | User working memory |
|
|
202
|
+
| Observational Memory | `mastra_observational_memory` | Observational memory generations |
|
|
203
|
+
| Workflows | `mastra_workflow_snapshots` | Workflow state |
|
|
204
|
+
| Scorers | `mastra_scorers` | Evaluation data |
|
|
205
|
+
| Cache | `mastra_cache` | Cache values, counters, and list metadata |
|
|
206
|
+
| Cache Items | `mastra_cache_list_items` | Cache list entries |
|
|
207
|
+
| Fallback | `mastra_documents` | Unknown tables |
|
|
208
|
+
|
|
209
|
+
### Observational memory
|
|
210
|
+
|
|
211
|
+
`ConvexStore` supports [observational memory](https://mastra.ai/docs/memory/observational-memory). Add `mastraObservationalMemoryTable` to your Convex schema and redeploy with `npx convex deploy` to enable it. Existing deployments created before this table was added need the same schema update.
|
|
205
212
|
|
|
206
213
|
### Architecture
|
|
207
214
|
|
|
@@ -228,7 +228,7 @@ const storage = new MongoDBStore({
|
|
|
228
228
|
})
|
|
229
229
|
```
|
|
230
230
|
|
|
231
|
-
> **Tip:** TTL indexes delete documents shortly after they expire (background thread runs every \~60 seconds), but the exact timing
|
|
231
|
+
> **Tip:** TTL indexes delete documents shortly after they expire (background thread runs every \~60 seconds), but the exact timing isn't guaranteed. For precise, immediate cleanup, use `prune()` instead.
|
|
232
232
|
|
|
233
233
|
## Reclaiming disk
|
|
234
234
|
|
|
@@ -100,7 +100,7 @@ const stream = await agent.stream('message for agent')
|
|
|
100
100
|
|
|
101
101
|
**options.structuredOutput.instructions** (`string`): Additional instructions for the structured output model.
|
|
102
102
|
|
|
103
|
-
**options.structuredOutput.jsonPromptInjection** (`boolean`):
|
|
103
|
+
**options.structuredOutput.jsonPromptInjection** (`boolean | 'system' | 'inline' | 'auto'`): Controls how the JSON schema reaches the model. Set to 'auto' to use native structured output when supported and inline prompt injection otherwise.
|
|
104
104
|
|
|
105
105
|
**options.structuredOutput.providerOptions** (`ProviderOptions`): Provider-specific options passed to the internal structuring agent. Use this to control model behavior like reasoning effort for thinking models (e.g., { openai: { reasoningEffort: 'low' } }).
|
|
106
106
|
|
|
@@ -124,6 +124,8 @@ const stream = await agent.stream('message for agent')
|
|
|
124
124
|
|
|
125
125
|
**options.memory.options** (`MemoryConfig`): Configuration for memory behavior including lastMessages, readOnly, semanticRecall, workingMemory, and filterIncompleteToolCalls.
|
|
126
126
|
|
|
127
|
+
**options.memory.onTitleGenerated** (`(title: string) => void | Promise<void>`): Callback fired asynchronously when a thread title is generated and persisted to storage. Title generation runs in the background and may complete after the stream ends. Only fires when generateTitle is enabled in memory options and the thread has no existing title.
|
|
128
|
+
|
|
127
129
|
**options.onFinish** (`StreamTextOnFinishCallback<any> | StreamObjectOnFinishCallback<OUTPUT>`): Callback function called when streaming completes. Receives the final result.
|
|
128
130
|
|
|
129
131
|
**options.onStepFinish** (`StreamTextOnStepFinishCallback<any> | never`): Callback function called after each execution step. Receives step details as a JSON string. Unavailable for structured output
|