@mastra/mcp-docs-server 1.3.0-alpha.6 → 1.3.1-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.docs/docs/evals/running-in-ci.md +7 -5
- package/.docs/docs/evals/vitest-integration.md +94 -31
- package/.docs/integrations/observability/opentelemetry.md +1 -0
- package/.docs/models/gateways/openrouter.md +2 -1
- package/.docs/models/index.md +1 -1
- package/.docs/models/providers/cortecs.md +5 -5
- package/.docs/models/providers/edenai.md +1 -1
- package/.docs/models/providers/kilo.md +7 -6
- package/.docs/models/providers/nano-gpt.md +1 -1
- package/.docs/reference/evals/run-evals.md +24 -18
- package/.docs/reference/evals/trajectory-accuracy.md +20 -16
- package/.docs/reference/observability/tracing/bridges/otel.md +12 -3
- package/.docs/reference/rag/vector-databases.md +23 -0
- package/.docs/reference/vectors/mongodb.md +90 -5
- package/dist/index.js +1 -1
- package/dist/{src-CGZ6-uLS.js → src-Baf8l9Sp.js} +17 -7
- package/dist/src-Baf8l9Sp.js.map +1 -0
- package/dist/stdio.js +1 -1
- package/dist/tools/embedded-docs.d.ts.map +1 -1
- package/package.json +6 -6
- package/dist/src-CGZ6-uLS.js.map +0 -1
|
@@ -14,11 +14,13 @@ You can use any testing framework that supports ESM modules, such as [Vitest](ht
|
|
|
14
14
|
|
|
15
15
|
Use `runEvals` to evaluate your agent against multiple test cases. The function accepts an array of data items, each containing an `input` and optional `groundTruth` for scorer validation.
|
|
16
16
|
|
|
17
|
+
Resolve the `target` with `mastra.getAgent()` or `mastra.getWorkflow()` rather than importing it directly. A directly imported agent or workflow has no Mastra instance attached, so registry lookups inside steps or tools (such as `mastra.getAgent('weatherAgent')`) fail, scores aren't persisted, and trace-based trajectory scoring is unavailable.
|
|
18
|
+
|
|
17
19
|
```typescript
|
|
18
20
|
import { describe, it, expect } from 'vitest'
|
|
19
|
-
import {
|
|
20
|
-
import {
|
|
21
|
-
import { locationScorer } from '../scorers/location-scorer'
|
|
21
|
+
import { runEvals } from '@mastra/core/evals'
|
|
22
|
+
import { mastra } from '../src/mastra'
|
|
23
|
+
import { locationScorer } from '../src/mastra/scorers/location-scorer'
|
|
22
24
|
|
|
23
25
|
describe('Weather Agent Tests', () => {
|
|
24
26
|
it('should correctly extract locations from queries', async () => {
|
|
@@ -37,7 +39,7 @@ describe('Weather Agent Tests', () => {
|
|
|
37
39
|
groundTruth: { expectedLocation: 'Berlin', expectedCountry: 'RU' },
|
|
38
40
|
},
|
|
39
41
|
],
|
|
40
|
-
target: weatherAgent,
|
|
42
|
+
target: mastra.getAgent('weatherAgent'),
|
|
41
43
|
scorers: [locationScorer],
|
|
42
44
|
})
|
|
43
45
|
|
|
@@ -73,7 +75,7 @@ Create separate test cases for different evaluation scenarios:
|
|
|
73
75
|
|
|
74
76
|
```typescript
|
|
75
77
|
describe('Weather Agent Tests', () => {
|
|
76
|
-
const
|
|
78
|
+
const weatherAgent = mastra.getAgent('weatherAgent')
|
|
77
79
|
|
|
78
80
|
it('should handle location disambiguation', async () => {
|
|
79
81
|
const result = await runEvals({
|
|
@@ -30,6 +30,29 @@ export default defineConfig({
|
|
|
30
30
|
|
|
31
31
|
The setup file registers the custom matchers on `expect`. Alternatively, call `registerEvalMatchers()` from `@mastra/evals/vitest` in your own setup file.
|
|
32
32
|
|
|
33
|
+
## Resolving targets from the Mastra instance
|
|
34
|
+
|
|
35
|
+
Get the `target` (and any registered scorers) from your `Mastra` instance with `mastra.getAgent()`, `mastra.getWorkflow()`, and `mastra.getScorer()` instead of importing the agent or workflow file directly. A directly imported agent or workflow has no Mastra instance attached, so anything that reaches into the registry at runtime fails. For example, a workflow step that calls `mastra.getAgent('weatherAgent')` throws because `mastra` is `undefined`, and the same applies to child workflows, `.agent('agent-id')` references by string, and tools that call `mastra.getWorkflow()`. Registered targets also attach the configured `storage`, which is what lets scores persist and trace-based trajectory scoring work. `runEvals` logs a warning when the target isn't registered.
|
|
36
|
+
|
|
37
|
+
The examples below look up `answerRelevancyScorer` and `trajectoryAccuracyScorer` with `mastra.getScorer()`, so register them on the `Mastra` instance first:
|
|
38
|
+
|
|
39
|
+
```typescript
|
|
40
|
+
import { Mastra } from '@mastra/core'
|
|
41
|
+
import {
|
|
42
|
+
createAnswerRelevancyScorer,
|
|
43
|
+
createTrajectoryAccuracyScorerCode,
|
|
44
|
+
} from '@mastra/evals/scorers/prebuilt'
|
|
45
|
+
|
|
46
|
+
export const mastra = new Mastra({
|
|
47
|
+
agents: { weatherAgent },
|
|
48
|
+
workflows: { weatherWorkflow },
|
|
49
|
+
scorers: {
|
|
50
|
+
answerRelevancyScorer: createAnswerRelevancyScorer({ model: 'openai/gpt-5-mini' }),
|
|
51
|
+
trajectoryAccuracyScorer: createTrajectoryAccuracyScorerCode(),
|
|
52
|
+
},
|
|
53
|
+
})
|
|
54
|
+
```
|
|
55
|
+
|
|
33
56
|
## Asserting on a dataset with `expectEvals`
|
|
34
57
|
|
|
35
58
|
`expectEvals` runs a `runEvals` evaluation inside a regular `test()` and asserts a minimum pass rate. It accepts the same configuration as `runEvals`: a `target` agent or workflow, `data` items, and `scorers`, `gates`, or thresholds. Gates score each item pass/fail, so `toPass(0.8)` requires at least 80% of items to pass every gate. Scorer thresholds still compare the average score across items and must pass regardless of the rate:
|
|
@@ -37,20 +60,15 @@ The setup file registers the custom matchers on `expect`. Alternatively, call `r
|
|
|
37
60
|
```typescript
|
|
38
61
|
import { test } from 'vitest'
|
|
39
62
|
import { expectEvals } from '@mastra/evals/vitest'
|
|
40
|
-
import {
|
|
41
|
-
import {
|
|
42
|
-
import { createKeywordCoverageScorer } from '@mastra/evals/scorers/prebuilt'
|
|
63
|
+
import { checks } from '@mastra/evals/checks'
|
|
64
|
+
import { mastra } from '../src/mastra'
|
|
43
65
|
|
|
44
|
-
test('
|
|
66
|
+
test('weather agent calls the tool and answers relevantly', { timeout: 60_000 }, async () => {
|
|
45
67
|
await expectEvals({
|
|
46
|
-
target:
|
|
47
|
-
data: [
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
{ input: 'What is the capital of Australia?', groundTruth: 'Canberra' },
|
|
51
|
-
],
|
|
52
|
-
gates: [containsGroundTruth],
|
|
53
|
-
scorers: [{ scorer: createKeywordCoverageScorer(), threshold: 0.4 }],
|
|
68
|
+
target: mastra.getAgent('weatherAgent'),
|
|
69
|
+
data: [{ input: "What's the weather in London?" }],
|
|
70
|
+
gates: [checks.calledTool('weatherTool'), checks.noToolErrors(), checks.includes('London')],
|
|
71
|
+
scorers: [{ scorer: mastra.getScorer('answerRelevancyScorer'), threshold: 0.7 }],
|
|
54
72
|
}).toPass(0.8)
|
|
55
73
|
})
|
|
56
74
|
```
|
|
@@ -66,18 +84,18 @@ LLM-backed evals are far slower than Vitest's default 5-second timeout, so pass
|
|
|
66
84
|
```typescript
|
|
67
85
|
import { test } from 'vitest'
|
|
68
86
|
import { expectEval } from '@mastra/evals/vitest'
|
|
69
|
-
import {
|
|
70
|
-
import {
|
|
87
|
+
import { checks } from '@mastra/evals/checks'
|
|
88
|
+
import { mastra } from '../src/mastra'
|
|
71
89
|
|
|
72
90
|
test.for([
|
|
73
|
-
{ input: '
|
|
74
|
-
{ input: '
|
|
75
|
-
{ input: '
|
|
76
|
-
])('
|
|
91
|
+
{ input: "What's the weather in London?", groundTruth: 'London' },
|
|
92
|
+
{ input: "What's the weather in Tokyo?", groundTruth: 'Tokyo' },
|
|
93
|
+
{ input: "What's the weather in Sydney?", groundTruth: 'Sydney' },
|
|
94
|
+
])('weather agent: $input', { timeout: 60_000 }, async item => {
|
|
77
95
|
await expectEval({
|
|
78
|
-
target:
|
|
96
|
+
target: mastra.getAgent('weatherAgent'),
|
|
79
97
|
data: item,
|
|
80
|
-
gates: [
|
|
98
|
+
gates: [checks.calledTool('weatherTool'), checks.includes(item.groundTruth)],
|
|
81
99
|
}).toPass()
|
|
82
100
|
})
|
|
83
101
|
```
|
|
@@ -91,20 +109,20 @@ For finer-grained control, call `runEvals` directly inside a regular `test()` an
|
|
|
91
109
|
```typescript
|
|
92
110
|
import { test, expect } from 'vitest'
|
|
93
111
|
import { runEvals } from '@mastra/core/evals'
|
|
94
|
-
import {
|
|
95
|
-
import {
|
|
112
|
+
import { checks } from '@mastra/evals/checks'
|
|
113
|
+
import { mastra } from '../src/mastra'
|
|
96
114
|
|
|
97
|
-
test('
|
|
115
|
+
test('weather agent quality', { timeout: 60_000 }, async () => {
|
|
98
116
|
const result = await runEvals({
|
|
99
|
-
target:
|
|
100
|
-
data: [{ input: '
|
|
101
|
-
scorers: [
|
|
102
|
-
gates: [
|
|
117
|
+
target: mastra.getAgent('weatherAgent'),
|
|
118
|
+
data: [{ input: "What's the weather in London?" }],
|
|
119
|
+
scorers: [mastra.getScorer('answerRelevancyScorer')],
|
|
120
|
+
gates: [checks.noToolErrors()],
|
|
103
121
|
})
|
|
104
122
|
|
|
105
123
|
expect(result).toHaveVerdict('passed')
|
|
106
124
|
expect(result).toPassGates()
|
|
107
|
-
expect(result).toHaveScoreAbove('relevancy', 0.7)
|
|
125
|
+
expect(result).toHaveScoreAbove('answer-relevancy-scorer', 0.7)
|
|
108
126
|
})
|
|
109
127
|
```
|
|
110
128
|
|
|
@@ -115,6 +133,49 @@ Available matchers:
|
|
|
115
133
|
- `toPassGates()`: asserts all gates passed. Fails when no gates were configured.
|
|
116
134
|
- `toPassThresholds()`: asserts all scorer thresholds passed. Fails when no thresholds were configured.
|
|
117
135
|
|
|
136
|
+
## Evaluating a workflow
|
|
137
|
+
|
|
138
|
+
Workflows are valid targets too. Use a `WorkflowScorerConfig` object for `scorers` to score the workflow output, individual steps, or the step execution trajectory, and reference nested scores with dot-paths in the matchers:
|
|
139
|
+
|
|
140
|
+
```typescript
|
|
141
|
+
import { test, expect } from 'vitest'
|
|
142
|
+
import { runEvals } from '@mastra/core/evals'
|
|
143
|
+
import { mastra } from '../src/mastra'
|
|
144
|
+
|
|
145
|
+
test('weather workflow runs fetch-weather then plan-activities', { timeout: 60_000 }, async () => {
|
|
146
|
+
const result = await runEvals({
|
|
147
|
+
target: mastra.getWorkflow('weatherWorkflow'),
|
|
148
|
+
data: [
|
|
149
|
+
{
|
|
150
|
+
input: { city: 'London' },
|
|
151
|
+
expectedTrajectory: {
|
|
152
|
+
steps: [
|
|
153
|
+
{ stepType: 'workflow_step', name: 'fetch-weather' },
|
|
154
|
+
{ stepType: 'workflow_step', name: 'plan-activities' },
|
|
155
|
+
],
|
|
156
|
+
},
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
input: { city: 'Tokyo' },
|
|
160
|
+
expectedTrajectory: {
|
|
161
|
+
steps: [
|
|
162
|
+
{ stepType: 'workflow_step', name: 'fetch-weather' },
|
|
163
|
+
{ stepType: 'workflow_step', name: 'plan-activities' },
|
|
164
|
+
],
|
|
165
|
+
},
|
|
166
|
+
},
|
|
167
|
+
],
|
|
168
|
+
scorers: {
|
|
169
|
+
trajectory: [mastra.getScorer('trajectoryAccuracyScorer')],
|
|
170
|
+
},
|
|
171
|
+
})
|
|
172
|
+
|
|
173
|
+
expect(result).toHaveScoreAbove('trajectory.code-trajectory-accuracy-scorer', 0.9)
|
|
174
|
+
})
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
The `plan-activities` step in this workflow calls `mastra.getAgent('weatherAgent')` at runtime, which only works because the target was resolved with `mastra.getWorkflow()`. See the [`runEvals` reference](https://mastra.ai/reference/evals/run-evals) for the full `WorkflowScorerConfig` shape.
|
|
178
|
+
|
|
118
179
|
## Reading the reporter output
|
|
119
180
|
|
|
120
181
|
`MastraEvalsReporter` prints a score table for every eval test after the run completes:
|
|
@@ -122,9 +183,11 @@ Available matchers:
|
|
|
122
183
|
```text
|
|
123
184
|
Mastra Evals
|
|
124
185
|
|
|
125
|
-
✓
|
|
126
|
-
|
|
127
|
-
|
|
186
|
+
✓ weather agent calls the tool and answers relevantly (1 item)
|
|
187
|
+
check-called-tool (gate) 1.0 ✓
|
|
188
|
+
check-no-tool-errors (gate) 1.0 ✓
|
|
189
|
+
check-includes (gate) 1.0 ✓
|
|
190
|
+
answer-relevancy-scorer (threshold: min 0.7) 0.9 ✓
|
|
128
191
|
|
|
129
192
|
Eval runs: 1 (1 passed)
|
|
130
193
|
```
|
|
@@ -861,6 +861,7 @@ Tags are exported as a JSON string in the `mastra.tags` span attribute for broad
|
|
|
861
861
|
If traces aren't displaying or connecting as expected:
|
|
862
862
|
|
|
863
863
|
- Verify OTEL SDK is initialized before Mastra (use the `--import` flag or import at the top of the entry point)
|
|
864
|
+
- If you see `[OtelBridge] No OpenTelemetry tracer provider is registered globally`, the bridge can't find a tracer provider. Some frameworks create a provider without registering it globally. Pass that provider directly with `new OtelBridge({ tracerProvider })`. See [setup requirements](https://mastra.ai/reference/observability/tracing/bridges/otel)
|
|
864
865
|
- Ensure the `OtelBridge` is added to your observability config
|
|
865
866
|
- Check that your OTEL backend is running and accessible
|
|
866
867
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# OpenRouter
|
|
6
6
|
|
|
7
|
-
OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access
|
|
7
|
+
OpenRouter aggregates models from multiple providers with enhanced features like rate limiting and failover. Access 385 models through Mastra's model router.
|
|
8
8
|
|
|
9
9
|
Learn more in the [OpenRouter documentation](https://openrouter.ai/models).
|
|
10
10
|
|
|
@@ -115,6 +115,7 @@ ANTHROPIC_API_KEY=ant-...
|
|
|
115
115
|
| `deepseek/deepseek-v4-pro-0813` |
|
|
116
116
|
| `deepseek/deepseek-v4.1-flash` |
|
|
117
117
|
| `dots-studio/dots-3-note-preview:free` |
|
|
118
|
+
| `fireworks/ember-1` |
|
|
118
119
|
| `google/gemini-2.5-flash` |
|
|
119
120
|
| `google/gemini-2.5-flash-image` |
|
|
120
121
|
| `google/gemini-2.5-flash-lite` |
|
package/.docs/models/index.md
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Model Providers
|
|
6
6
|
|
|
7
|
-
Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to
|
|
7
|
+
Mastra provides a unified interface for working with LLMs across multiple providers, giving you access to 7616 models from 210 providers through a single API.
|
|
8
8
|
|
|
9
9
|
## Features
|
|
10
10
|
|
|
@@ -53,7 +53,7 @@ for await (const chunk of stream) {
|
|
|
53
53
|
| `cortecs/codestral-2508` | 256K | | | | | | $0.37 | $1 |
|
|
54
54
|
| `cortecs/deepseek-r1-0528` | 164K | | | | | | $0.65 | $3 |
|
|
55
55
|
| `cortecs/deepseek-v3.2` | 164K | | | | | | $0.30 | $0.49 |
|
|
56
|
-
| `cortecs/deepseek-v4-flash-0731` | 1.0M | | | | | | $0.
|
|
56
|
+
| `cortecs/deepseek-v4-flash-0731` | 1.0M | | | | | | $0.09 | $0.17 |
|
|
57
57
|
| `cortecs/deepseek-v4-pro` | 1.0M | | | | | | $2 | $3 |
|
|
58
58
|
| `cortecs/deepseek-v4-pro-0813` | 1.0M | | | | | | $2 | $4 |
|
|
59
59
|
| `cortecs/deepseek-v4.1-flash` | 1.0M | | | | | | $0.50 | $1 |
|
|
@@ -73,8 +73,8 @@ for await (const chunk of stream) {
|
|
|
73
73
|
| `cortecs/glm-5` | 203K | | | | | | $0.99 | $3 |
|
|
74
74
|
| `cortecs/glm-5-turbo` | 203K | | | | | | $1 | $4 |
|
|
75
75
|
| `cortecs/glm-5.1` | 203K | | | | | | $1 | $4 |
|
|
76
|
-
| `cortecs/glm-5.2` | 1.0M | | | | | | $
|
|
77
|
-
| `cortecs/glm-5.3` | 1.0M | | | | | | $
|
|
76
|
+
| `cortecs/glm-5.2` | 1.0M | | | | | | $0.90 | $3 |
|
|
77
|
+
| `cortecs/glm-5.3` | 1.0M | | | | | | $0.87 | $3 |
|
|
78
78
|
| `cortecs/glm-5.3-flash` | 1.0M | | | | | | $0.10 | $0.35 |
|
|
79
79
|
| `cortecs/glm-5v-turbo` | 203K | | | | | | $1 | $4 |
|
|
80
80
|
| `cortecs/gpt-4.1` | 1.0M | | | | | | $2 | $9 |
|
|
@@ -97,8 +97,8 @@ for await (const chunk of stream) {
|
|
|
97
97
|
| `cortecs/gpt-oss-safeguard-120b` | 128K | | | | | | $0.18 | $0.70 |
|
|
98
98
|
| `cortecs/hermes-4-405b` | 128K | | | | | | $1.00 | $3 |
|
|
99
99
|
| `cortecs/kimi-k2.5` | 262K | | | | | | $0.49 | $3 |
|
|
100
|
-
| `cortecs/kimi-k2.6` | 262K | | | | | | $0.
|
|
101
|
-
| `cortecs/kimi-k2.7-code` | 262K | | | | | | $0.
|
|
100
|
+
| `cortecs/kimi-k2.6` | 262K | | | | | | $0.47 | $3 |
|
|
101
|
+
| `cortecs/kimi-k2.7-code` | 262K | | | | | | $0.66 | $3 |
|
|
102
102
|
| `cortecs/kimi-k3` | 1.0M | | | | | | $3 | $15 |
|
|
103
103
|
| `cortecs/llama-3.1-8b-instruct` | 128K | | | | | | $0.17 | $0.17 |
|
|
104
104
|
| `cortecs/llama-3.3-70b-instruct` | 131K | | | | | | $0.72 | $0.72 |
|
|
@@ -72,7 +72,7 @@ for await (const chunk of stream) {
|
|
|
72
72
|
| `edenai/anthropic/claude-opus-4-8` | 1.0M | | | | | | $5 | $25 |
|
|
73
73
|
| `edenai/anthropic/claude-opus-5` | 1.0M | | | | | | $5 | $25 |
|
|
74
74
|
| `edenai/anthropic/claude-opus-5-5` | 1.0M | | | | | | $4 | $20 |
|
|
75
|
-
| `edenai/anthropic/claude-opus-latest` | 1.0M | | | | | | $
|
|
75
|
+
| `edenai/anthropic/claude-opus-latest` | 1.0M | | | | | | $4 | $20 |
|
|
76
76
|
| `edenai/anthropic/claude-sonnet-4-6` | 1.0M | | | | | | $3 | $15 |
|
|
77
77
|
| `edenai/anthropic/claude-sonnet-5` | 1.0M | | | | | | $2 | $10 |
|
|
78
78
|
| `edenai/anthropic/claude-sonnet-latest` | 1.0M | | | | | | $2 | $10 |
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
# Kilo Gateway
|
|
6
6
|
|
|
7
|
-
Access
|
|
7
|
+
Access 392 Kilo Gateway models through Mastra's model router. Authentication is handled automatically using the `KILO_API_KEY` environment variable.
|
|
8
8
|
|
|
9
9
|
Learn more in the [Kilo Gateway documentation](https://kilo.ai).
|
|
10
10
|
|
|
@@ -42,9 +42,9 @@ for await (const chunk of stream) {
|
|
|
42
42
|
| `kilo/~anthropic/claude-haiku-latest` | 200K | | | | | | $1 | $5 |
|
|
43
43
|
| `kilo/~anthropic/claude-opus-latest` | 1.0M | | | | | | $4 | $20 |
|
|
44
44
|
| `kilo/~anthropic/claude-sonnet-latest` | 1.0M | | | | | | $2 | $10 |
|
|
45
|
-
| `kilo/~deepseek/deepseek-flash-latest` | 1.0M | | | | | | $0.
|
|
45
|
+
| `kilo/~deepseek/deepseek-flash-latest` | 1.0M | | | | | | $0.04 | $1 |
|
|
46
46
|
| `kilo/~deepseek/deepseek-pro-latest` | 1.0M | | | | | | $0.39 | $3 |
|
|
47
|
-
| `kilo/~deepseek/deepseek-v4-flash-latest` | 1.0M | | | | | | $0.
|
|
47
|
+
| `kilo/~deepseek/deepseek-v4-flash-latest` | 1.0M | | | | | | $0.03 | $0.32 |
|
|
48
48
|
| `kilo/~google/gemini-flash-latest` | 1.0M | | | | | | $0.75 | $4 |
|
|
49
49
|
| `kilo/~google/gemini-pro-latest` | 1.0M | | | | | | $2 | $12 |
|
|
50
50
|
| `kilo/~moonshotai/kimi-latest` | 1.0M | | | | | | $1 | $11 |
|
|
@@ -115,6 +115,7 @@ for await (const chunk of stream) {
|
|
|
115
115
|
| `kilo/deepseek/deepseek-v4-pro-0813` | 1.0M | | | | | | $1 | $4 |
|
|
116
116
|
| `kilo/deepseek/deepseek-v4.1-flash` | 1.0M | | | | | | $0.30 | $1 |
|
|
117
117
|
| `kilo/dots-studio/dots-3-note-preview:free` | 512K | | | | | | — | — |
|
|
118
|
+
| `kilo/fireworks/ember-1` | 1.0M | | | | | | $3 | $15 |
|
|
118
119
|
| `kilo/google/gemini-2.5-flash` | 1.0M | | | | | | $0.30 | $3 |
|
|
119
120
|
| `kilo/google/gemini-2.5-flash-image` | 33K | | | | | | $0.15 | $1 |
|
|
120
121
|
| `kilo/google/gemini-2.5-flash-lite` | 1.0M | | | | | | $0.10 | $0.40 |
|
|
@@ -210,7 +211,7 @@ for await (const chunk of stream) {
|
|
|
210
211
|
| `kilo/moonshotai/kimi-k2-thinking` | 262K | | | | | | $0.60 | $3 |
|
|
211
212
|
| `kilo/moonshotai/kimi-k2.5` | 262K | | | | | | $0.60 | $3 |
|
|
212
213
|
| `kilo/moonshotai/kimi-k2.6` | 262K | | | | | | $0.95 | $4 |
|
|
213
|
-
| `kilo/moonshotai/kimi-k2.7-code` | 262K | | | | | | $0.
|
|
214
|
+
| `kilo/moonshotai/kimi-k2.7-code` | 262K | | | | | | $0.66 | $3 |
|
|
214
215
|
| `kilo/moonshotai/kimi-k3` | 1.0M | | | | | | $3 | $15 |
|
|
215
216
|
| `kilo/morph/morph-v3-fast` | 82K | | | | | | $0.80 | $1 |
|
|
216
217
|
| `kilo/morph/morph-v3-large` | 262K | | | | | | $0.90 | $2 |
|
|
@@ -318,7 +319,7 @@ for await (const chunk of stream) {
|
|
|
318
319
|
| `kilo/qwen/qwen3-235b-a22b-2507` | 262K | | | | | | $0.15 | $0.60 |
|
|
319
320
|
| `kilo/qwen/qwen3-235b-a22b-thinking-2507` | 131K | | | | | | $0.23 | $2 |
|
|
320
321
|
| `kilo/qwen/qwen3-30b-a3b` | 41K | | | | | | $0.13 | $0.52 |
|
|
321
|
-
| `kilo/qwen/qwen3-30b-a3b-instruct-2507` |
|
|
322
|
+
| `kilo/qwen/qwen3-30b-a3b-instruct-2507` | 262K | | | | | | $0.13 | $0.52 |
|
|
322
323
|
| `kilo/qwen/qwen3-30b-a3b-thinking-2507` | 82K | | | | | | $0.20 | $2 |
|
|
323
324
|
| `kilo/qwen/qwen3-32b` | 41K | | | | | | $0.08 | $0.28 |
|
|
324
325
|
| `kilo/qwen/qwen3-8b` | 131K | | | | | | $0.12 | $0.46 |
|
|
@@ -383,7 +384,7 @@ for await (const chunk of stream) {
|
|
|
383
384
|
| `kilo/stepfun/step-3.7-flash:free` | 262K | | | | | | — | — |
|
|
384
385
|
| `kilo/tencent/hunyuan-a13b-instruct` | 131K | | | | | | $0.14 | $0.57 |
|
|
385
386
|
| `kilo/tencent/hy-mt2-1.8b` | 8K | | | | | | $0.04 | $0.18 |
|
|
386
|
-
| `kilo/tencent/hy-mt2-30b-a3b` | 8K | | | | | | $0.07 | $0.
|
|
387
|
+
| `kilo/tencent/hy-mt2-30b-a3b` | 8K | | | | | | $0.07 | $0.29 |
|
|
387
388
|
| `kilo/tencent/hy-mt2-7b` | 8K | | | | | | $0.07 | $0.29 |
|
|
388
389
|
| `kilo/tencent/hy3` | 262K | | | | | | $0.13 | $0.53 |
|
|
389
390
|
| `kilo/tencent/hy3-preview` | 262K | | | | | | $0.18 | $0.60 |
|
|
@@ -266,6 +266,7 @@ for await (const chunk of stream) {
|
|
|
266
266
|
| `nano-gpt/inference-net/schematron-v2-turbo` | 128K | | | | | | $0.03 | $0.15 |
|
|
267
267
|
| `nano-gpt/inflatebot/MN-12B-Mag-Mell-R1` | 16K | | | | | | $0.49 | $0.49 |
|
|
268
268
|
| `nano-gpt/kimi-k2-instruct-fast` | 131K | | | | | | $0.40 | $2 |
|
|
269
|
+
| `nano-gpt/kitani/clover-1-150b` | 200K | | | | | | $0.15 | $0.80 |
|
|
269
270
|
| `nano-gpt/LatitudeGames/Wayfarer-Large-70B-Llama-3.3` | 33K | | | | | | $0.70 | $0.70 |
|
|
270
271
|
| `nano-gpt/lightonai/LightOnOCR-2-1B` | 33K | | | | | | $0.18 | $0.35 |
|
|
271
272
|
| `nano-gpt/liquid/lfm-2.5-2.6b` | 128K | | | | | | $0.10 | $0.20 |
|
|
@@ -339,7 +340,6 @@ for await (const chunk of stream) {
|
|
|
339
340
|
| `nano-gpt/moonshotai/kimi-k3` | 1.0M | | | | | | $2 | $10 |
|
|
340
341
|
| `nano-gpt/moonshotai/kimi-latest` | 1.0M | | | | | | $2 | $10 |
|
|
341
342
|
| `nano-gpt/nano-gpt-help` | 6K | | | | | | — | — |
|
|
342
|
-
| `nano-gpt/nano/lumen-stealth` | 200K | | | | | | $0.05 | — |
|
|
343
343
|
| `nano-gpt/nanogpt/coding-router` | 1.0M | | | | | | $1 | $2 |
|
|
344
344
|
| `nano-gpt/nanogpt/coding-router:high` | 1.0M | | | | | | $1 | $2 |
|
|
345
345
|
| `nano-gpt/nanogpt/coding-router:low` | 1.0M | | | | | | $0.14 | $0.28 |
|
|
@@ -10,17 +10,16 @@ The `runEvals` function enables batch evaluation of agents and workflows by runn
|
|
|
10
10
|
|
|
11
11
|
```typescript
|
|
12
12
|
import { runEvals } from '@mastra/core/evals'
|
|
13
|
-
import {
|
|
14
|
-
import { myScorer1, myScorer2 } from './scorers'
|
|
13
|
+
import { mastra } from './mastra'
|
|
15
14
|
|
|
16
15
|
const result = await runEvals({
|
|
17
|
-
target: myAgent,
|
|
16
|
+
target: mastra.getAgent('myAgent'),
|
|
18
17
|
data: [
|
|
19
18
|
{ input: 'What is machine learning?' },
|
|
20
19
|
{ input: 'Explain neural networks' },
|
|
21
20
|
{ input: 'How does AI work?' },
|
|
22
21
|
],
|
|
23
|
-
scorers: [myScorer1, myScorer2],
|
|
22
|
+
scorers: [mastra.getScorer('myScorer1'), mastra.getScorer('myScorer2')],
|
|
24
23
|
targetOptions: { maxSteps: 5 },
|
|
25
24
|
concurrency: 2,
|
|
26
25
|
onItemComplete: ({ item, targetResult, scorerResults }) => {
|
|
@@ -33,15 +32,17 @@ console.log(`Average scores:`, result.scores)
|
|
|
33
32
|
console.log(`Processed ${result.summary.totalItems} items`)
|
|
34
33
|
```
|
|
35
34
|
|
|
35
|
+
Resolve the `target` from the `Mastra` instance with `mastra.getAgent()` or `mastra.getWorkflow()` rather than importing the agent or workflow directly. A directly imported target has no Mastra instance attached, so registry lookups made during execution (a step calling `mastra.getAgent()`, a child workflow, an `.agent('agent-id')` reference, a tool calling `mastra.getWorkflow()`) fail, scores aren't persisted, and trace-based trajectory extraction is unavailable. `runEvals` logs a warning when the target isn't registered.
|
|
36
|
+
|
|
36
37
|
### Multi-turn evaluation
|
|
37
38
|
|
|
38
39
|
```typescript
|
|
39
40
|
import { runEvals } from '@mastra/core/evals'
|
|
40
41
|
import { checks } from '@mastra/evals/checks'
|
|
41
|
-
import {
|
|
42
|
+
import { mastra } from './mastra'
|
|
42
43
|
|
|
43
44
|
const result = await runEvals({
|
|
44
|
-
target: weatherAgent,
|
|
45
|
+
target: mastra.getAgent('weatherAgent'),
|
|
45
46
|
data: [
|
|
46
47
|
{
|
|
47
48
|
inputs: [
|
|
@@ -60,13 +61,16 @@ const result = await runEvals({
|
|
|
60
61
|
```typescript
|
|
61
62
|
import { runEvals } from '@mastra/core/evals'
|
|
62
63
|
import { checks } from '@mastra/evals/checks'
|
|
63
|
-
import {
|
|
64
|
+
import { mastra } from './mastra'
|
|
64
65
|
|
|
65
66
|
const result = await runEvals({
|
|
66
|
-
target:
|
|
67
|
+
target: mastra.getAgent('weatherAgent'),
|
|
67
68
|
data: [{ input: 'What is the weather in Brooklyn?' }],
|
|
68
69
|
gates: [checks.calledTool('get_weather'), checks.noToolErrors()],
|
|
69
|
-
scorers: [
|
|
70
|
+
scorers: [
|
|
71
|
+
{ scorer: mastra.getScorer('faithfulnessScorer'), threshold: 0.7 },
|
|
72
|
+
checks.includes('Brooklyn'),
|
|
73
|
+
],
|
|
70
74
|
})
|
|
71
75
|
|
|
72
76
|
result.verdict // 'passed' | 'scored' | 'failed'
|
|
@@ -173,7 +177,7 @@ import { runEvals } from '@mastra/core/evals'
|
|
|
173
177
|
import { checks } from '@mastra/evals/checks'
|
|
174
178
|
|
|
175
179
|
const result = await runEvals({
|
|
176
|
-
target: weatherAgent,
|
|
180
|
+
target: mastra.getAgent('weatherAgent'),
|
|
177
181
|
data: [{ input: 'What is the weather in Brooklyn?' }],
|
|
178
182
|
gates: [checks.calledTool('get_weather'), checks.noToolErrors()],
|
|
179
183
|
scorers: [
|
|
@@ -213,7 +217,7 @@ const myScorer = createScorer({
|
|
|
213
217
|
})
|
|
214
218
|
|
|
215
219
|
const result = await runEvals({
|
|
216
|
-
target: chatAgent,
|
|
220
|
+
target: mastra.getAgent('chatAgent'),
|
|
217
221
|
data: [
|
|
218
222
|
{
|
|
219
223
|
input: 'What is AI?',
|
|
@@ -240,7 +244,7 @@ import { createTrajectoryAccuracyScorerCode } from '@mastra/evals/scorers/code/t
|
|
|
240
244
|
const trajectoryScorer = createTrajectoryAccuracyScorerCode()
|
|
241
245
|
|
|
242
246
|
const result = await runEvals({
|
|
243
|
-
target: chatAgent,
|
|
247
|
+
target: mastra.getAgent('chatAgent'),
|
|
244
248
|
data: [
|
|
245
249
|
{
|
|
246
250
|
input: 'What is the weather in London?',
|
|
@@ -265,7 +269,7 @@ Pass execution options like `maxSteps` or `modelSettings` to customize agent beh
|
|
|
265
269
|
|
|
266
270
|
```typescript
|
|
267
271
|
const result = await runEvals({
|
|
268
|
-
target: chatAgent,
|
|
272
|
+
target: mastra.getAgent('chatAgent'),
|
|
269
273
|
data: [{ input: 'Summarize this article' }, { input: 'Translate to French' }],
|
|
270
274
|
scorers: [relevancyScorer],
|
|
271
275
|
targetOptions: {
|
|
@@ -277,9 +281,11 @@ const result = await runEvals({
|
|
|
277
281
|
|
|
278
282
|
### Workflow Evaluation
|
|
279
283
|
|
|
284
|
+
Resolve the workflow with `mastra.getWorkflow()` so steps that call `mastra.getAgent()` or run child workflows can reach the registry:
|
|
285
|
+
|
|
280
286
|
```typescript
|
|
281
287
|
const workflowResult = await runEvals({
|
|
282
|
-
target: myWorkflow,
|
|
288
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
283
289
|
data: [
|
|
284
290
|
{ input: { query: 'Process this data', priority: 'high' } },
|
|
285
291
|
{ input: { query: 'Another task', priority: 'low' } },
|
|
@@ -309,7 +315,7 @@ Add trajectory scoring to workflow evaluations to validate step execution order:
|
|
|
309
315
|
|
|
310
316
|
```typescript
|
|
311
317
|
const workflowResult = await runEvals({
|
|
312
|
-
target: myWorkflow,
|
|
318
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
313
319
|
data: [
|
|
314
320
|
{
|
|
315
321
|
input: { query: 'Process this data' },
|
|
@@ -340,7 +346,7 @@ Use `startOptions` on individual data items to customize each workflow run. Per-
|
|
|
340
346
|
|
|
341
347
|
```typescript
|
|
342
348
|
const result = await runEvals({
|
|
343
|
-
target: myWorkflow,
|
|
349
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
344
350
|
data: [
|
|
345
351
|
{
|
|
346
352
|
input: { query: 'hello' },
|
|
@@ -362,7 +368,7 @@ Use `inputs` to send sequential turns on a shared thread. Scorers see the accumu
|
|
|
362
368
|
|
|
363
369
|
```typescript
|
|
364
370
|
const result = await runEvals({
|
|
365
|
-
target: chatAgent,
|
|
371
|
+
target: mastra.getAgent('chatAgent'),
|
|
366
372
|
data: [
|
|
367
373
|
{
|
|
368
374
|
inputs: ['My favorite city is Brooklyn.', 'What is the weather in my favorite city?'],
|
|
@@ -385,7 +391,7 @@ Use `turns` to attach `gates`/`scorers` to individual turns. Each per-turn asser
|
|
|
385
391
|
|
|
386
392
|
```typescript
|
|
387
393
|
const result = await runEvals({
|
|
388
|
-
target: chatAgent,
|
|
394
|
+
target: mastra.getAgent('chatAgent'),
|
|
389
395
|
data: [
|
|
390
396
|
{
|
|
391
397
|
turns: [
|
|
@@ -221,7 +221,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
221
221
|
const scorer = createTrajectoryAccuracyScorerCode()
|
|
222
222
|
|
|
223
223
|
await runEvals({
|
|
224
|
-
target: myAgent,
|
|
224
|
+
target: mastra.getAgent('myAgent'),
|
|
225
225
|
scorers: { trajectory: [scorer] },
|
|
226
226
|
data: [
|
|
227
227
|
{
|
|
@@ -245,16 +245,20 @@ await runEvals({
|
|
|
245
245
|
|
|
246
246
|
### Evaluation modes
|
|
247
247
|
|
|
248
|
-
The code-based scorer operates in
|
|
248
|
+
The code-based scorer operates in one of three modes based on `ordering`:
|
|
249
249
|
|
|
250
|
-
#### Strict mode (`
|
|
250
|
+
#### Strict mode (`ordering: 'strict'`)
|
|
251
251
|
|
|
252
|
-
|
|
252
|
+
Only an exact match (the same steps in the same order, with nothing extra or missing) scores `1.0`. Anything else gets partial credit for the expected steps that matched in position, with a penalty deducted for each extra step. For example, the two expected steps followed by one extra step scores `0.75`.
|
|
253
253
|
|
|
254
|
-
#### Relaxed mode (`
|
|
254
|
+
#### Relaxed mode (`ordering: 'relaxed'`, default)
|
|
255
255
|
|
|
256
256
|
Allows extra steps. Expected steps must appear in the correct relative order. The score is calculated based on how many expected steps were matched, with optional penalties for extra or repeated steps.
|
|
257
257
|
|
|
258
|
+
#### Unordered mode (`ordering: 'unordered'`)
|
|
259
|
+
|
|
260
|
+
Only checks that each expected step is present. Order is ignored, and extra steps are reported in the result but not penalized.
|
|
261
|
+
|
|
258
262
|
## Code-based scoring details
|
|
259
263
|
|
|
260
264
|
- **Continuous scores**: Returns values between 0.0 and 1.0 in relaxed mode; binary (0 or 1) in strict mode
|
|
@@ -303,16 +307,16 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
303
307
|
{ stepType: 'tool_call', name: 'fetch-tool' },
|
|
304
308
|
],
|
|
305
309
|
},
|
|
306
|
-
comparisonOptions: {
|
|
310
|
+
comparisonOptions: { ordering: 'strict' },
|
|
307
311
|
})
|
|
308
312
|
|
|
309
313
|
const result = await runEvals({
|
|
310
|
-
target: myAgent,
|
|
314
|
+
target: mastra.getAgent('myAgent'),
|
|
311
315
|
scorers: { trajectory: [scorer] },
|
|
312
316
|
data: [{ input: 'Get my data' }],
|
|
313
317
|
})
|
|
314
318
|
|
|
315
|
-
console.log(result.scores.trajectory['trajectory-accuracy']) // 1.0
|
|
319
|
+
console.log(result.scores.trajectory['code-trajectory-accuracy-scorer']) // 1.0
|
|
316
320
|
```
|
|
317
321
|
|
|
318
322
|
### Agent trajectory with relaxed ordering
|
|
@@ -327,7 +331,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
327
331
|
{ stepType: 'tool_call', name: 'summarize-tool' },
|
|
328
332
|
],
|
|
329
333
|
},
|
|
330
|
-
comparisonOptions: {
|
|
334
|
+
comparisonOptions: { ordering: 'relaxed' },
|
|
331
335
|
})
|
|
332
336
|
|
|
333
337
|
// Agent called search-tool → log-tool → summarize-tool
|
|
@@ -354,12 +358,12 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
354
358
|
})
|
|
355
359
|
|
|
356
360
|
const result = await runEvals({
|
|
357
|
-
target: myWorkflow,
|
|
361
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
358
362
|
scorers: { trajectory: [scorer] },
|
|
359
363
|
data: [{ input: { data: 'test' } }],
|
|
360
364
|
})
|
|
361
365
|
|
|
362
|
-
console.log(result.scores.trajectory['trajectory-accuracy'])
|
|
366
|
+
console.log(result.scores.trajectory['code-trajectory-accuracy-scorer'])
|
|
363
367
|
```
|
|
364
368
|
|
|
365
369
|
### Comparing step data
|
|
@@ -517,7 +521,7 @@ const scorer = createTrajectoryScorerCode({
|
|
|
517
521
|
})
|
|
518
522
|
|
|
519
523
|
const result = await runEvals({
|
|
520
|
-
target: myAgent,
|
|
524
|
+
target: mastra.getAgent('myAgent'),
|
|
521
525
|
scorers: { trajectory: [scorer] },
|
|
522
526
|
data: [
|
|
523
527
|
{
|
|
@@ -583,7 +587,7 @@ const trajectoryScorer = createTrajectoryAccuracyScorerCode({
|
|
|
583
587
|
})
|
|
584
588
|
|
|
585
589
|
const result = await runEvals({
|
|
586
|
-
target: myAgent,
|
|
590
|
+
target: mastra.getAgent('myAgent'),
|
|
587
591
|
scorers: {
|
|
588
592
|
agent: [qualityScorer], // receives raw MastraDBMessage[] output
|
|
589
593
|
trajectory: [trajectoryScorer], // receives pre-extracted Trajectory
|
|
@@ -592,7 +596,7 @@ const result = await runEvals({
|
|
|
592
596
|
})
|
|
593
597
|
|
|
594
598
|
// result.scores.agent['quality'] — agent-level score
|
|
595
|
-
// result.scores.trajectory['trajectory-accuracy'] — trajectory score
|
|
599
|
+
// result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
|
|
596
600
|
```
|
|
597
601
|
|
|
598
602
|
### Workflow trajectory evaluation
|
|
@@ -612,7 +616,7 @@ const workflowTrajectoryScorer = createTrajectoryAccuracyScorerCode({
|
|
|
612
616
|
})
|
|
613
617
|
|
|
614
618
|
const result = await runEvals({
|
|
615
|
-
target: myWorkflow,
|
|
619
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
616
620
|
scorers: {
|
|
617
621
|
workflow: [outputScorer], // receives workflow output
|
|
618
622
|
trajectory: [workflowTrajectoryScorer], // receives pre-extracted Trajectory from step results
|
|
@@ -621,7 +625,7 @@ const result = await runEvals({
|
|
|
621
625
|
})
|
|
622
626
|
|
|
623
627
|
// result.scores.workflow['output-quality'] — workflow-level score
|
|
624
|
-
// result.scores.trajectory['trajectory-accuracy'] — trajectory score
|
|
628
|
+
// result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
|
|
625
629
|
```
|
|
626
630
|
|
|
627
631
|
## Related
|