@mastra/mcp-docs-server 1.2.7 → 1.2.8-alpha.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.docs/docs/agent-builder/deploying.md +1 -0
- package/.docs/docs/agent-builder/integrations.md +1 -3
- package/.docs/docs/agent-builder/overview.md +1 -1
- package/.docs/docs/agent-controller/overview.md +1 -1
- package/.docs/docs/agents/overview.md +17 -13
- package/.docs/docs/agents/skills.md +1 -3
- package/.docs/docs/agents/structured-output.md +14 -7
- package/.docs/docs/agents/using-tools.md +80 -26
- package/.docs/docs/browser/agent-browser.md +1 -0
- package/.docs/docs/browser/firecrawl.md +129 -0
- package/.docs/docs/browser/overview.md +16 -2
- package/.docs/docs/capabilities/channels/other-adapters.md +1 -1
- package/.docs/docs/capabilities/channels/overview.md +1 -0
- package/.docs/docs/capabilities/channels/slack.md +113 -3
- package/.docs/docs/deployment/mastra-server.md +35 -0
- package/.docs/docs/deployment/monorepo.md +2 -0
- package/.docs/docs/deployment/overview.md +6 -0
- package/.docs/docs/deployment/sandbox.md +277 -0
- package/.docs/docs/editor/overview.md +2 -6
- package/.docs/docs/editor/tools.md +2 -6
- package/.docs/docs/evals/multi-turn.md +178 -0
- package/.docs/docs/evals/overview.md +1 -0
- package/.docs/docs/evals/running-in-ci.md +3 -9
- package/.docs/docs/getting-started/file-based-agents.md +2 -2
- package/.docs/docs/index.md +17 -13
- package/.docs/docs/long-running-agents/background-tasks.md +2 -2
- package/.docs/docs/long-running-agents/durable-agents.md +2 -1
- package/.docs/docs/long-running-agents/schedules.md +2 -2
- package/.docs/docs/mastra-platform/database.md +13 -6
- package/.docs/docs/mastra-platform/deploy.md +17 -8
- package/.docs/docs/mastra-platform/environments.md +5 -4
- package/.docs/docs/mastra-platform/regions.md +73 -0
- package/.docs/docs/mcp/overview.md +1 -1
- package/.docs/docs/memory/multi-user-threads.md +4 -4
- package/.docs/docs/memory/observational-memory.md +78 -3
- package/.docs/docs/memory/overview.md +4 -0
- package/.docs/docs/memory/working-memory.md +1 -1
- package/.docs/docs/observability/integrations/bridges/otel.md +1 -3
- package/.docs/docs/observability/integrations/exporters/braintrust.md +34 -2
- package/.docs/docs/observability/integrations/exporters/otel.md +1 -3
- package/.docs/docs/server/auth/simple-auth.md +1 -3
- package/.docs/docs/server/request-context.md +2 -0
- package/.docs/docs/studio/overview.md +2 -0
- package/.docs/docs/voice/livekit.md +1 -1
- package/.docs/docs/workflows/control-flow.md +1 -3
- package/.docs/guides/build-your-ui/ai-sdk-ui.md +1 -0
- package/.docs/guides/guide/signal-provider.md +2 -0
- package/.docs/guides/migrations/vnext-to-standard-apis.md +1 -3
- package/.docs/models/environment-variables.md +1 -0
- package/.docs/models/gateways/openrouter.md +339 -345
- package/.docs/models/gateways/vercel.md +6 -5
- package/.docs/models/index.md +1 -1
- package/.docs/models/providers/ambient.md +12 -10
- package/.docs/models/providers/baseten.md +3 -2
- package/.docs/models/providers/crossmodel.md +2 -1
- package/.docs/models/providers/deepinfra.md +4 -7
- package/.docs/models/providers/empiriolabs.md +2 -1
- package/.docs/models/providers/evroc.md +3 -2
- package/.docs/models/providers/google.md +2 -2
- package/.docs/models/providers/kilo.md +1 -1
- package/.docs/models/providers/kimi-for-coding.md +10 -11
- package/.docs/models/providers/llmgateway.md +3 -8
- package/.docs/models/providers/moonshotai-cn.md +3 -2
- package/.docs/models/providers/moonshotai.md +3 -2
- package/.docs/models/providers/nebius.md +3 -1
- package/.docs/models/providers/ollama-cloud.md +25 -42
- package/.docs/models/providers/opencode-go.md +5 -3
- package/.docs/models/providers/orcarouter.md +1 -1
- package/.docs/models/providers/privatemode-ai.md +10 -10
- package/.docs/models/providers/thinkingmachines.md +73 -0
- package/.docs/models/providers/togetherai.md +2 -6
- package/.docs/models/providers/zenmux.md +2 -1
- package/.docs/models/providers.md +1 -0
- package/.docs/reference/acp/acp-agent.md +2 -0
- package/.docs/reference/acp/create-acp-tool.md +28 -0
- package/.docs/reference/agent-controller/agent-controller-class.md +1 -1
- package/.docs/reference/agent-controller/session.md +15 -0
- package/.docs/reference/agents/agent.md +14 -2
- package/.docs/reference/agents/durable-agent.md +23 -0
- package/.docs/reference/agents/generate.md +19 -1
- package/.docs/reference/browser/firecrawl-browser.md +71 -0
- package/.docs/reference/cli/mastra.md +14 -6
- package/.docs/reference/code-sdk/mount-agent-controller.md +1 -3
- package/.docs/reference/core/getMCPServer.md +1 -3
- package/.docs/reference/core/getMCPServerById.md +1 -3
- package/.docs/reference/core/getTool.md +33 -0
- package/.docs/reference/core/getToolById.md +48 -0
- package/.docs/reference/core/getWorkflow.md +1 -3
- package/.docs/reference/core/listMCPServers.md +2 -6
- package/.docs/reference/core/listTools.md +38 -0
- package/.docs/reference/core/mastra-class.md +1 -1
- package/.docs/reference/datasets/addItem.md +1 -3
- package/.docs/reference/datasets/addItems.md +1 -3
- package/.docs/reference/datasets/compareExperiments.md +1 -3
- package/.docs/reference/datasets/create.md +1 -3
- package/.docs/reference/datasets/dataset.md +2 -6
- package/.docs/reference/datasets/datasets-manager.md +5 -15
- package/.docs/reference/datasets/delete.md +1 -3
- package/.docs/reference/datasets/deleteExperiment.md +1 -3
- package/.docs/reference/datasets/deleteItem.md +1 -3
- package/.docs/reference/datasets/deleteItems.md +1 -3
- package/.docs/reference/datasets/get.md +1 -3
- package/.docs/reference/datasets/getDetails.md +1 -3
- package/.docs/reference/datasets/getExperiment.md +1 -3
- package/.docs/reference/datasets/getItem.md +1 -3
- package/.docs/reference/datasets/getItemHistory.md +1 -3
- package/.docs/reference/datasets/list.md +1 -3
- package/.docs/reference/datasets/listExperimentResults.md +1 -3
- package/.docs/reference/datasets/listExperiments.md +1 -3
- package/.docs/reference/datasets/listItems.md +1 -3
- package/.docs/reference/datasets/listVersions.md +1 -3
- package/.docs/reference/datasets/startExperiment.md +1 -3
- package/.docs/reference/datasets/startExperimentAsync.md +1 -3
- package/.docs/reference/datasets/update.md +1 -3
- package/.docs/reference/datasets/updateItem.md +1 -3
- package/.docs/reference/editor/mastra-editor.md +1 -3
- package/.docs/reference/evals/create-scorer.md +4 -10
- package/.docs/reference/evals/mastra-scorer.md +1 -3
- package/.docs/reference/evals/run-evals.md +94 -2
- package/.docs/reference/file-based-agents/logger.md +1 -1
- package/.docs/reference/file-based-agents/tools.md +7 -6
- package/.docs/reference/index.md +5 -0
- package/.docs/reference/memory/observational-memory.md +21 -0
- package/.docs/reference/observability/tracing/processors/sensitive-data-filter.md +1 -3
- package/.docs/reference/processors/token-limiter-processor.md +1 -3
- package/.docs/reference/rag/database-config.md +12 -0
- package/.docs/reference/storage/clickhouse.md +1 -1
- package/.docs/reference/storage/convex.md +17 -10
- package/.docs/reference/storage/retention.md +1 -1
- package/.docs/reference/streaming/agents/stream.md +3 -1
- package/.docs/reference/tools/create-tool.md +78 -52
- package/.docs/reference/tools/mcp-client.md +104 -24
- package/.docs/reference/tools/mcp-server.md +143 -44
- package/.docs/reference/tools/task-tools.md +2 -2
- package/.docs/reference/tools/tavily.md +1 -1
- package/.docs/reference/tools/vector-query-tool.md +22 -0
- package/.docs/reference/vectors/mongodb.md +2 -0
- package/.docs/reference/vectors/turbopuffer.md +4 -0
- package/.docs/reference/voice/mistral.md +127 -0
- package/.docs/reference/workspace/daytona-sandbox.md +2 -6
- package/CHANGELOG.md +64 -0
- package/dist/{chunk-GLPCVXXO.js → chunk-HGADBLKG.js} +2 -2
- package/dist/{chunk-GLPCVXXO.js.map → chunk-HGADBLKG.js.map} +1 -1
- package/dist/index.js +1 -1
- package/dist/stdio.js +1 -1
- package/dist/tools/docs.d.ts.map +1 -1
- package/package.json +7 -7
|
@@ -12,18 +12,14 @@ const server1 = new MCPServer({
|
|
|
12
12
|
id: 'server-one',
|
|
13
13
|
name: 'Server One',
|
|
14
14
|
version: '1.0.0',
|
|
15
|
-
tools: {
|
|
16
|
-
/* ... */
|
|
17
|
-
},
|
|
15
|
+
tools: {/* ... */},
|
|
18
16
|
})
|
|
19
17
|
|
|
20
18
|
const server2 = new MCPServer({
|
|
21
19
|
id: 'server-two',
|
|
22
20
|
name: 'Server Two',
|
|
23
21
|
version: '1.0.0',
|
|
24
|
-
tools: {
|
|
25
|
-
/* ... */
|
|
26
|
-
},
|
|
22
|
+
tools: {/* ... */},
|
|
27
23
|
})
|
|
28
24
|
|
|
29
25
|
export const mastra = new Mastra({
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
> Discover all available pages from the documentation index: https://mastra.ai/llms.txt
|
|
2
|
+
|
|
3
|
+
# Mastra.listTools()
|
|
4
|
+
|
|
5
|
+
The `.listTools()` method returns the tools configured on a Mastra instance. The returned record uses Mastra registration keys and tool instances as values.
|
|
6
|
+
|
|
7
|
+
## Usage example
|
|
8
|
+
|
|
9
|
+
```typescript
|
|
10
|
+
import { Mastra } from '@mastra/core/mastra'
|
|
11
|
+
import { weatherTool } from './tools/weather-tool'
|
|
12
|
+
|
|
13
|
+
export const mastra = new Mastra({
|
|
14
|
+
tools: {
|
|
15
|
+
weather: weatherTool,
|
|
16
|
+
},
|
|
17
|
+
})
|
|
18
|
+
|
|
19
|
+
const tools = mastra.listTools()
|
|
20
|
+
const weather = tools?.weather
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Parameters
|
|
24
|
+
|
|
25
|
+
This method doesn't accept any parameters.
|
|
26
|
+
|
|
27
|
+
## Returns
|
|
28
|
+
|
|
29
|
+
**tools** (`TTools | undefined`): The Mastra-level tool registry, where each key is a registration key and each value is a tool instance. Returns undefined when no registry is available.
|
|
30
|
+
|
|
31
|
+
This method lists the Mastra-level registry. It doesn't include tools that exist only in a function-based Agent `tools` callback.
|
|
32
|
+
|
|
33
|
+
## Related
|
|
34
|
+
|
|
35
|
+
- [Share tools across agents](https://mastra.ai/docs/agents/using-tools)
|
|
36
|
+
- [Mastra class](https://mastra.ai/reference/core/mastra-class)
|
|
37
|
+
- [Mastra.getTool()](https://mastra.ai/reference/core/getTool)
|
|
38
|
+
- [Mastra.getToolById()](https://mastra.ai/reference/core/getToolById)
|
|
@@ -53,7 +53,7 @@ Visit the [Configuration reference](https://mastra.ai/reference/configuration) f
|
|
|
53
53
|
|
|
54
54
|
**agents** (`Record<string, Agent>`): Agent instances to register, keyed by name (Default: `{}`)
|
|
55
55
|
|
|
56
|
-
**tools** (`Record<string, ToolApi>`):
|
|
56
|
+
**tools** (`Record<string, ToolApi>`): Tool instances to register. Keys are registration keys used by \`getTool()\`, and values are tool instances. Use \`getToolById()\` for intrinsic ID lookup and \`listTools()\` to read the registry. (Default: `{}`)
|
|
57
57
|
|
|
58
58
|
**storage** (`MastraCompositeStore`): Storage engine instance for persisting data
|
|
59
59
|
|
|
@@ -11,9 +11,7 @@ Adds a single item to the dataset. Each item has an input, optional ground truth
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Adds multiple items to the dataset in a single bulk operation.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Compares two or more experiments, producing per-item and per-scorer comparisons.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const comparison = await mastra.datasets.compareExperiments({
|
|
19
17
|
experimentIds: ['exp-baseline', 'exp-new'],
|
|
@@ -12,9 +12,7 @@ Creates a new dataset. Use Standard JSON Schema (Zod, Valibot, ArkType, etc.) fo
|
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
import { z } from 'zod'
|
|
14
14
|
|
|
15
|
-
const mastra = new Mastra({
|
|
16
|
-
/* storage config */
|
|
17
|
-
})
|
|
15
|
+
const mastra = new Mastra({/* storage config */})
|
|
18
16
|
|
|
19
17
|
// Create with metadata
|
|
20
18
|
const dataset = await mastra.datasets.create({
|
|
@@ -13,9 +13,7 @@ The `Dataset` class provides methods for item CRUD, versioning, and experiment m
|
|
|
13
13
|
```typescript
|
|
14
14
|
import { Mastra } from '@mastra/core'
|
|
15
15
|
|
|
16
|
-
const mastra = new Mastra({
|
|
17
|
-
/* storage config */
|
|
18
|
-
})
|
|
16
|
+
const mastra = new Mastra({/* storage config */})
|
|
19
17
|
|
|
20
18
|
const dataset = await mastra.datasets.create({
|
|
21
19
|
name: 'QA pairs',
|
|
@@ -44,9 +42,7 @@ console.log(`${summary.succeededCount}/${summary.totalItems} succeeded`)
|
|
|
44
42
|
```typescript
|
|
45
43
|
import { Mastra } from '@mastra/core'
|
|
46
44
|
|
|
47
|
-
const mastra = new Mastra({
|
|
48
|
-
/* storage config */
|
|
49
|
-
})
|
|
45
|
+
const mastra = new Mastra({/* storage config */})
|
|
50
46
|
|
|
51
47
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
52
48
|
|
|
@@ -13,9 +13,7 @@ The `DatasetsManager` class provides the public API for managing datasets, inclu
|
|
|
13
13
|
```typescript
|
|
14
14
|
import { Mastra } from '@mastra/core'
|
|
15
15
|
|
|
16
|
-
const mastra = new Mastra({
|
|
17
|
-
/* storage config */
|
|
18
|
-
})
|
|
16
|
+
const mastra = new Mastra({/* storage config */})
|
|
19
17
|
|
|
20
18
|
// Create a dataset
|
|
21
19
|
const dataset = await mastra.datasets.create({
|
|
@@ -32,9 +30,7 @@ const { datasets } = await mastra.datasets.list()
|
|
|
32
30
|
```typescript
|
|
33
31
|
import { Mastra } from '@mastra/core'
|
|
34
32
|
|
|
35
|
-
const mastra = new Mastra({
|
|
36
|
-
/* storage config */
|
|
37
|
-
})
|
|
33
|
+
const mastra = new Mastra({/* storage config */})
|
|
38
34
|
|
|
39
35
|
// Get by ID
|
|
40
36
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
@@ -50,9 +46,7 @@ Retrieves a specific experiment (run) by ID. Unlike `dataset.getExperiment()`, t
|
|
|
50
46
|
```typescript
|
|
51
47
|
import { Mastra } from '@mastra/core'
|
|
52
48
|
|
|
53
|
-
const mastra = new Mastra({
|
|
54
|
-
/* storage config */
|
|
55
|
-
})
|
|
49
|
+
const mastra = new Mastra({/* storage config */})
|
|
56
50
|
|
|
57
51
|
// Get experiment directly without knowing the dataset
|
|
58
52
|
const experiment = await mastra.datasets.getExperiment({
|
|
@@ -68,9 +62,7 @@ console.log(`Status: ${experiment.status}`)
|
|
|
68
62
|
```typescript
|
|
69
63
|
import { Mastra } from '@mastra/core'
|
|
70
64
|
|
|
71
|
-
const mastra = new Mastra({
|
|
72
|
-
/* storage config */
|
|
73
|
-
})
|
|
65
|
+
const mastra = new Mastra({/* storage config */})
|
|
74
66
|
|
|
75
67
|
const comparison = await mastra.datasets.compareExperiments({
|
|
76
68
|
experimentIds: ['exp-1', 'exp-2'],
|
|
@@ -83,9 +75,7 @@ const comparison = await mastra.datasets.compareExperiments({
|
|
|
83
75
|
`DatasetsManager` isn't instantiated directly. Access it from a `Mastra` instance:
|
|
84
76
|
|
|
85
77
|
```typescript
|
|
86
|
-
const mastra = new Mastra({
|
|
87
|
-
/* storage config */
|
|
88
|
-
})
|
|
78
|
+
const mastra = new Mastra({/* storage config */})
|
|
89
79
|
const datasetsManager = mastra.datasets
|
|
90
80
|
```
|
|
91
81
|
|
|
@@ -11,9 +11,7 @@ Deletes a dataset by ID, including all items, versions, and experiments.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
await mastra.datasets.delete({ id: 'dataset-id' })
|
|
19
17
|
```
|
|
@@ -11,9 +11,7 @@ Deletes an experiment (run) by ID, including all associated results.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Deletes a single item from the dataset by ID.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Deletes multiple items from the dataset in a single bulk operation.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Retrieves an existing dataset by ID. Throws a `MastraError` if the dataset doesn
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Retrieves the full dataset record from storage, including metadata, schemas, and
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Retrieves a specific experiment (run) by its ID.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Retrieves a single dataset item by ID, optionally at a specific dataset version.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Retrieves the full SCD-2 (Slowly Changing Dimension Type 2) history of a specifi
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Lists datasets with pagination and optional filters. Filters are applied at the
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const { datasets, pagination } = await mastra.datasets.list({ page: 0, perPage: 10 })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Lists individual item results for a specific experiment with optional filters an
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Lists experiments (runs) for this dataset with optional filters and pagination.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Lists items in the dataset. When only `version` is provided, returns a bare `Dat
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Lists all versions of the dataset with pagination.
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Runs an experiment on the dataset and waits for completion. Executes all items a
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -11,9 +11,7 @@ Starts an experiment asynchronously (fire-and-forget). Returns immediately with
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -12,9 +12,7 @@ Updates dataset metadata, name, description, and/or schemas. Use Standard JSON S
|
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
import { z } from 'zod'
|
|
14
14
|
|
|
15
|
-
const mastra = new Mastra({
|
|
16
|
-
/* storage config */
|
|
17
|
-
})
|
|
15
|
+
const mastra = new Mastra({/* storage config */})
|
|
18
16
|
|
|
19
17
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
20
18
|
|
|
@@ -11,9 +11,7 @@ Updates an existing item in the dataset. Only the provided fields are updated. U
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { Mastra } from '@mastra/core'
|
|
13
13
|
|
|
14
|
-
const mastra = new Mastra({
|
|
15
|
-
/* storage config */
|
|
16
|
-
})
|
|
14
|
+
const mastra = new Mastra({/* storage config */})
|
|
17
15
|
|
|
18
16
|
const dataset = await mastra.datasets.get({ id: 'dataset-id' })
|
|
19
17
|
|
|
@@ -14,9 +14,7 @@ import { MastraEditor } from '@mastra/editor'
|
|
|
14
14
|
import { ComposioToolProvider } from '@mastra/editor/providers/composio'
|
|
15
15
|
|
|
16
16
|
export const mastra = new Mastra({
|
|
17
|
-
agents: {
|
|
18
|
-
/* your agents */
|
|
19
|
-
},
|
|
17
|
+
agents: {/* your agents */},
|
|
20
18
|
editor: new MastraEditor({
|
|
21
19
|
toolProviders: {
|
|
22
20
|
composio: new ComposioToolProvider({
|
|
@@ -23,18 +23,12 @@ const scorer = createScorer({
|
|
|
23
23
|
instructions: 'You are an expert evaluator...',
|
|
24
24
|
},
|
|
25
25
|
})
|
|
26
|
-
.preprocess({
|
|
27
|
-
|
|
28
|
-
})
|
|
29
|
-
.analyze({
|
|
30
|
-
/* step config */
|
|
31
|
-
})
|
|
26
|
+
.preprocess({/* step config */})
|
|
27
|
+
.analyze({/* step config */})
|
|
32
28
|
.generateScore(({ run, results }) => {
|
|
33
29
|
// Return a number
|
|
34
30
|
})
|
|
35
|
-
.generateReason({
|
|
36
|
-
/* step config */
|
|
37
|
-
})
|
|
31
|
+
.generateReason({/* step config */})
|
|
38
32
|
```
|
|
39
33
|
|
|
40
34
|
## `createScorer` options
|
|
@@ -51,7 +45,7 @@ const scorer = createScorer({
|
|
|
51
45
|
|
|
52
46
|
**judge.instructions** (`string`): System prompt/instructions for the LLM.
|
|
53
47
|
|
|
54
|
-
**judge.jsonPromptInjection** (`boolean
|
|
48
|
+
**judge.jsonPromptInjection** (`boolean | 'system' | 'inline' | 'auto'`): Controls how the judge's structured output schema reaches the model. Defaults to 'auto', which uses native structured output when supported and inline prompt injection otherwise. Explicit values override automatic routing.
|
|
55
49
|
|
|
56
50
|
**judge.inputProcessors** (`Processor[]`): Input processors applied to the internal judge agent before its messages reach the model (e.g. redaction, validation).
|
|
57
51
|
|
|
@@ -33,9 +33,7 @@ const result = await scorer.run({
|
|
|
33
33
|
input: 'What is machine learning?',
|
|
34
34
|
output: 'Machine learning is a subset of artificial intelligence...',
|
|
35
35
|
runId: 'optional-run-id',
|
|
36
|
-
requestContext: {
|
|
37
|
-
/* optional context */
|
|
38
|
-
},
|
|
36
|
+
requestContext: {/* optional context */},
|
|
39
37
|
})
|
|
40
38
|
```
|
|
41
39
|
|
|
@@ -31,6 +31,28 @@ console.log(`Average scores:`, result.scores)
|
|
|
31
31
|
console.log(`Processed ${result.summary.totalItems} items`)
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
+
### Multi-turn evaluation
|
|
35
|
+
|
|
36
|
+
```typescript
|
|
37
|
+
import { runEvals } from '@mastra/core/evals'
|
|
38
|
+
import { checks } from '@mastra/evals/checks'
|
|
39
|
+
import { weatherAgent } from './agents/weather-agent'
|
|
40
|
+
|
|
41
|
+
const result = await runEvals({
|
|
42
|
+
target: weatherAgent,
|
|
43
|
+
data: [
|
|
44
|
+
{
|
|
45
|
+
inputs: [
|
|
46
|
+
'What is the weather in Brooklyn?',
|
|
47
|
+
'What about tomorrow?',
|
|
48
|
+
'Compare the two forecasts.',
|
|
49
|
+
],
|
|
50
|
+
},
|
|
51
|
+
],
|
|
52
|
+
scorers: [checks.calledTool('get_weather', { times: 2 }), checks.includes('Brooklyn')],
|
|
53
|
+
})
|
|
54
|
+
```
|
|
55
|
+
|
|
34
56
|
### With gates and thresholds
|
|
35
57
|
|
|
36
58
|
```typescript
|
|
@@ -60,7 +82,7 @@ result.thresholdResults // [{ id, passed, averageScore, threshold }]
|
|
|
60
82
|
|
|
61
83
|
**gates** (`MastraScorer[]`): Scorers that must score 1.0 for the run to pass. If any gate averages below 1.0 across data items, the verdict is failed. Gates run before regular scorers on each data item. When provided, scorers may be omitted.
|
|
62
84
|
|
|
63
|
-
**targetOptions** (`AgentExecutionOptions | WorkflowRunOptions`): Options forwarded to the target during execution. For agents: options passed to agent.generate() (e.g. maxSteps, modelSettings, instructions). For workflows: options passed to run.start() (e.g. perStep, outputOptions, initialState).
|
|
85
|
+
**targetOptions** (`AgentExecutionOptions | WorkflowRunOptions`): Options forwarded to the target during execution. For agents: options passed to agent.generate() (e.g. maxSteps, modelSettings, instructions). For workflows: options passed to run.start() (e.g. perStep, outputOptions, initialState). For multi-turn agent runs (inputs/turns), runEvals generates and injects the shared thread and a resource, so memory.thread is optional; provide memory.resource to reuse a specific resource.
|
|
64
86
|
|
|
65
87
|
**concurrency** (`number`): Number of test cases to run concurrently. (Default: `1`)
|
|
66
88
|
|
|
@@ -68,7 +90,11 @@ result.thresholdResults // [{ id, passed, averageScore, threshold }]
|
|
|
68
90
|
|
|
69
91
|
## Data item structure
|
|
70
92
|
|
|
71
|
-
**input** (`string | string[] | CoreMessage[] | any`): Input data for the target. For agents: messages or strings. For workflows: workflow input data.
|
|
93
|
+
**input** (`string | string[] | CoreMessage[] | any`): Input data for the target. For agents: messages or strings. For workflows: workflow input data. Optional when inputs is provided.
|
|
94
|
+
|
|
95
|
+
**inputs** (`(string | string[] | CoreMessage[] | any)[]`): Multi-turn inputs. Each entry is one turn (same shape as input) sent sequentially to the agent on the same thread. Scorers see the accumulated output from all turns. Only supported for Agent targets. When provided, input can be omitted. Mutually exclusive with turns.
|
|
96
|
+
|
|
97
|
+
**turns** (`EvalTurn[]`): Multi-turn conversation with per-turn assertions. Each turn is an object { input, gates?, scorers? } sent sequentially on the same thread; its gates/scorers evaluate only that turn's input and output. Per-turn outcomes are reported in turnResults and folded into the overall verdict. Only supported for Agent targets. Mutually exclusive with input and inputs.
|
|
72
98
|
|
|
73
99
|
**groundTruth** (`any`): Expected or reference output for comparison during scoring.
|
|
74
100
|
|
|
@@ -112,6 +138,18 @@ For workflows, use `WorkflowScorerConfig` to specify scorers at different levels
|
|
|
112
138
|
|
|
113
139
|
**thresholdResults** (`ThresholdResult[]`): Per-threshold-scorer results averaged across all data items. Each entry has id, passed, averageScore, and threshold.
|
|
114
140
|
|
|
141
|
+
**turnResults** (`TurnResult[]`): Present when any data item uses turns. Each entry has index (zero-based turn), optional gateResults, thresholdResults, and scores (bare-scorer averages keyed by scorer id), aggregated by turn index across data items.
|
|
142
|
+
|
|
143
|
+
## EvalTurn
|
|
144
|
+
|
|
145
|
+
A single turn in a `turns` array. Its `gates`/`scorers` evaluate only that turn's input and output:
|
|
146
|
+
|
|
147
|
+
**input** (`string | string[] | CoreMessage[] | any`): The input sent to the agent for this turn.
|
|
148
|
+
|
|
149
|
+
**gates** (`MastraScorer[]`): Gates that must score 1.0 for this turn. A failing turn gate makes the overall verdict failed.
|
|
150
|
+
|
|
151
|
+
**scorers** (`ScorerEntry[]`): Scorers (optionally with thresholds) evaluated against this turn only. A missed per-turn threshold (with gates passing) makes the verdict scored.
|
|
152
|
+
|
|
115
153
|
## ScorerEntry
|
|
116
154
|
|
|
117
155
|
A scorer entry in the `scorers` array can be either a bare scorer or a scorer with a threshold:
|
|
@@ -314,8 +352,62 @@ const result = await runEvals({
|
|
|
314
352
|
})
|
|
315
353
|
```
|
|
316
354
|
|
|
355
|
+
### Multi-turn conversation evaluation
|
|
356
|
+
|
|
357
|
+
Use `inputs` to send sequential turns on a shared thread. Scorers see the accumulated output from all turns:
|
|
358
|
+
|
|
359
|
+
```typescript
|
|
360
|
+
const result = await runEvals({
|
|
361
|
+
target: chatAgent,
|
|
362
|
+
data: [
|
|
363
|
+
{
|
|
364
|
+
inputs: ['My favorite city is Brooklyn.', 'What is the weather in my favorite city?'],
|
|
365
|
+
},
|
|
366
|
+
],
|
|
367
|
+
gates: [checks.calledTool('get_weather')],
|
|
368
|
+
scorers: [{ scorer: checks.similarity('Brooklyn weather forecast'), threshold: 0.5 }],
|
|
369
|
+
})
|
|
370
|
+
|
|
371
|
+
// result.verdict: 'passed' | 'scored' | 'failed'
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
Each turn runs `agent.generate()` with the same `threadId`, so the agent sees the full conversation history. `runEvals` also injects a `resourceId` (Mastra memory scopes messages by resource + thread), defaulting it to the generated thread; pass `targetOptions.memory.resource` to pin a specific one. Cross-turn recall requires the agent to have a memory store configured; otherwise turns run in isolation. Mix single-turn (`input`) and multi-turn (`inputs`) items in the same `data` array. When using `inputs`, `input` can be omitted.
|
|
375
|
+
|
|
376
|
+
Scoring uses the accumulated output from all turns as `run.output`, but only the first turn as `run.input`. Prefer output-based scorers (`checks.includes`, `checks.calledTool`, `checks.similarity`) for multi-turn; input-relative scorers (e.g. faithfulness) only see the first turn's input. Trajectory scorers that read from the trace (`AgentScorerConfig.trajectory`) resolve against the last turn's span; tool-call checks that read `run.output` (like `checks.calledTool`) still see every turn.
|
|
377
|
+
|
|
378
|
+
### Per-turn assertions
|
|
379
|
+
|
|
380
|
+
Use `turns` to attach `gates`/`scorers` to individual turns. Each per-turn assertion sees only that turn's input and output, so a later-turn regression can't be hidden by an earlier turn:
|
|
381
|
+
|
|
382
|
+
```typescript
|
|
383
|
+
const result = await runEvals({
|
|
384
|
+
target: chatAgent,
|
|
385
|
+
data: [
|
|
386
|
+
{
|
|
387
|
+
turns: [
|
|
388
|
+
{
|
|
389
|
+
input: 'What is the weather in Brooklyn?',
|
|
390
|
+
gates: [checks.calledTool('get_weather')],
|
|
391
|
+
},
|
|
392
|
+
{
|
|
393
|
+
input: 'What about tomorrow?',
|
|
394
|
+
gates: [checks.calledTool('get_weather')], // must call again this turn
|
|
395
|
+
scorers: [{ scorer: checks.similarity('tomorrow forecast'), threshold: 0.5 }],
|
|
396
|
+
},
|
|
397
|
+
],
|
|
398
|
+
},
|
|
399
|
+
],
|
|
400
|
+
})
|
|
401
|
+
|
|
402
|
+
result.verdict // folds in per-turn gate/threshold outcomes
|
|
403
|
+
result.turnResults // [{ index, gateResults, thresholdResults, scores }]
|
|
404
|
+
```
|
|
405
|
+
|
|
406
|
+
Per-turn gates/scorers evaluate only that turn (`run.input`/`run.output` are that turn's). A failing turn gate makes the verdict `failed`; a missed turn threshold (gates passing) makes it `scored`. Top-level `scorers`/`gates` still score the accumulated conversation holistically. `turns` is Agent-only and cannot be combined with `input` or `inputs`.
|
|
407
|
+
|
|
317
408
|
## Related
|
|
318
409
|
|
|
410
|
+
- [Multi-turn evals](https://mastra.ai/docs/evals/multi-turn): Conceptual guide to multi-turn evaluation
|
|
319
411
|
- [Gates and Verdicts](https://mastra.ai/docs/evals/gates-and-verdicts): Conceptual guide to severity semantics
|
|
320
412
|
- [Quick Checks](https://mastra.ai/reference/evals/checks): Zero-LLM composable micro-scorers
|
|
321
413
|
- [createScorer()](https://mastra.ai/reference/evals/create-scorer): Create custom scorers for experiments
|
|
@@ -19,7 +19,7 @@ export default new PinoLogger({
|
|
|
19
19
|
})
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
Mastra registers the logger before storage, observability, and file-based agents, so those primitives log through this logger as they
|
|
22
|
+
Mastra registers the logger before storage, observability, and file-based agents, so those primitives log through this logger as they're wired up.
|
|
23
23
|
|
|
24
24
|
## Precedence with code
|
|
25
25
|
|
|
@@ -16,16 +16,17 @@ import { z } from 'zod'
|
|
|
16
16
|
|
|
17
17
|
export default createTool({
|
|
18
18
|
id: 'get_weather',
|
|
19
|
-
description: 'Get the current weather for a
|
|
19
|
+
description: 'Get the current weather for a location.',
|
|
20
20
|
inputSchema: z.object({
|
|
21
|
-
|
|
21
|
+
location: z.string(),
|
|
22
22
|
}),
|
|
23
23
|
outputSchema: z.object({
|
|
24
|
-
|
|
25
|
-
|
|
24
|
+
location: z.string(),
|
|
25
|
+
temperatureCelsius: z.number(),
|
|
26
|
+
conditions: z.string(),
|
|
26
27
|
}),
|
|
27
|
-
execute: async ({
|
|
28
|
-
return {
|
|
28
|
+
execute: async ({ location }) => {
|
|
29
|
+
return { location, temperatureCelsius: 21, conditions: 'sunny' }
|
|
29
30
|
},
|
|
30
31
|
})
|
|
31
32
|
```
|