@mastra/mcp-docs-server 1.3.0 → 1.3.1-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,11 +14,13 @@ You can use any testing framework that supports ESM modules, such as [Vitest](ht
14
14
 
15
15
  Use `runEvals` to evaluate your agent against multiple test cases. The function accepts an array of data items, each containing an `input` and optional `groundTruth` for scorer validation.
16
16
 
17
+ Resolve the `target` with `mastra.getAgent()` or `mastra.getWorkflow()` rather than importing it directly. A directly imported agent or workflow has no Mastra instance attached, so registry lookups inside steps or tools (such as `mastra.getAgent('weatherAgent')`) fail, scores aren't persisted, and trace-based trajectory scoring is unavailable.
18
+
17
19
  ```typescript
18
20
  import { describe, it, expect } from 'vitest'
19
- import { createScorer, runEvals } from '@mastra/core/evals'
20
- import { weatherAgent } from './weather-agent'
21
- import { locationScorer } from '../scorers/location-scorer'
21
+ import { runEvals } from '@mastra/core/evals'
22
+ import { mastra } from '../src/mastra'
23
+ import { locationScorer } from '../src/mastra/scorers/location-scorer'
22
24
 
23
25
  describe('Weather Agent Tests', () => {
24
26
  it('should correctly extract locations from queries', async () => {
@@ -37,7 +39,7 @@ describe('Weather Agent Tests', () => {
37
39
  groundTruth: { expectedLocation: 'Berlin', expectedCountry: 'RU' },
38
40
  },
39
41
  ],
40
- target: weatherAgent,
42
+ target: mastra.getAgent('weatherAgent'),
41
43
  scorers: [locationScorer],
42
44
  })
43
45
 
@@ -73,7 +75,7 @@ Create separate test cases for different evaluation scenarios:
73
75
 
74
76
  ```typescript
75
77
  describe('Weather Agent Tests', () => {
76
- const locationScorer = createScorer({/* ... */})
78
+ const weatherAgent = mastra.getAgent('weatherAgent')
77
79
 
78
80
  it('should handle location disambiguation', async () => {
79
81
  const result = await runEvals({
@@ -37,20 +37,15 @@ The setup file registers the custom matchers on `expect`. Alternatively, call `r
37
37
  ```typescript
38
38
  import { test } from 'vitest'
39
39
  import { expectEvals } from '@mastra/evals/vitest'
40
- import { capitalsAgent } from './capitals-agent'
41
- import { containsGroundTruth } from '../scorers'
42
- import { createKeywordCoverageScorer } from '@mastra/evals/scorers/prebuilt'
40
+ import { checks } from '@mastra/evals/checks'
41
+ import { mastra } from '../src/mastra'
43
42
 
44
- test('capitals agent answers with the expected city', { timeout: 60_000 }, async () => {
43
+ test('weather agent calls the tool and answers relevantly', { timeout: 60_000 }, async () => {
45
44
  await expectEvals({
46
- target: capitalsAgent,
47
- data: [
48
- { input: 'What is the capital of France?', groundTruth: 'Paris' },
49
- { input: 'What is the capital of Japan?', groundTruth: 'Tokyo' },
50
- { input: 'What is the capital of Australia?', groundTruth: 'Canberra' },
51
- ],
52
- gates: [containsGroundTruth],
53
- scorers: [{ scorer: createKeywordCoverageScorer(), threshold: 0.4 }],
45
+ target: mastra.getAgent('weatherAgent'),
46
+ data: [{ input: "What's the weather in London?" }],
47
+ gates: [checks.calledTool('weatherTool'), checks.noToolErrors(), checks.includes('London')],
48
+ scorers: [{ scorer: mastra.getScorer('answerRelevancyScorer'), threshold: 0.7 }],
54
49
  }).toPass(0.8)
55
50
  })
56
51
  ```
@@ -66,18 +61,18 @@ LLM-backed evals are far slower than Vitest's default 5-second timeout, so pass
66
61
  ```typescript
67
62
  import { test } from 'vitest'
68
63
  import { expectEval } from '@mastra/evals/vitest'
69
- import { capitalsAgent } from './capitals-agent'
70
- import { containsGroundTruth } from '../scorers'
64
+ import { checks } from '@mastra/evals/checks'
65
+ import { mastra } from '../src/mastra'
71
66
 
72
67
  test.for([
73
- { input: 'What is the capital of France?', groundTruth: 'Paris' },
74
- { input: 'What is the capital of Japan?', groundTruth: 'Tokyo' },
75
- { input: 'What is the capital of Australia?', groundTruth: 'Canberra' },
76
- ])('capitals agent: $input', { timeout: 60_000 }, async item => {
68
+ { input: "What's the weather in London?", groundTruth: 'London' },
69
+ { input: "What's the weather in Tokyo?", groundTruth: 'Tokyo' },
70
+ { input: "What's the weather in Sydney?", groundTruth: 'Sydney' },
71
+ ])('weather agent: $input', { timeout: 60_000 }, async item => {
77
72
  await expectEval({
78
- target: capitalsAgent,
73
+ target: mastra.getAgent('weatherAgent'),
79
74
  data: item,
80
- gates: [containsGroundTruth],
75
+ gates: [checks.calledTool('weatherTool'), checks.includes(item.groundTruth)],
81
76
  }).toPass()
82
77
  })
83
78
  ```
@@ -91,20 +86,20 @@ For finer-grained control, call `runEvals` directly inside a regular `test()` an
91
86
  ```typescript
92
87
  import { test, expect } from 'vitest'
93
88
  import { runEvals } from '@mastra/core/evals'
94
- import { supportAgent } from './support-agent'
95
- import { relevancyScorer, noRefusalScorer } from '../scorers'
89
+ import { checks } from '@mastra/evals/checks'
90
+ import { mastra } from '../src/mastra'
96
91
 
97
- test('support agent quality', { timeout: 60_000 }, async () => {
92
+ test('weather agent quality', { timeout: 60_000 }, async () => {
98
93
  const result = await runEvals({
99
- target: supportAgent,
100
- data: [{ input: 'How do I update my payment method?' }],
101
- scorers: [relevancyScorer],
102
- gates: [noRefusalScorer],
94
+ target: mastra.getAgent('weatherAgent'),
95
+ data: [{ input: "What's the weather in London?" }],
96
+ scorers: [mastra.getScorer('answerRelevancyScorer')],
97
+ gates: [checks.noToolErrors()],
103
98
  })
104
99
 
105
100
  expect(result).toHaveVerdict('passed')
106
101
  expect(result).toPassGates()
107
- expect(result).toHaveScoreAbove('relevancy', 0.7)
102
+ expect(result).toHaveScoreAbove('answer-relevancy-scorer', 0.7)
108
103
  })
109
104
  ```
110
105
 
@@ -115,6 +110,49 @@ Available matchers:
115
110
  - `toPassGates()`: asserts all gates passed. Fails when no gates were configured.
116
111
  - `toPassThresholds()`: asserts all scorer thresholds passed. Fails when no thresholds were configured.
117
112
 
113
+ ## Evaluating a workflow
114
+
115
+ Workflows are valid targets too. Use a `WorkflowScorerConfig` object for `scorers` to score the workflow output, individual steps, or the step execution trajectory, and reference nested scores with dot-paths in the matchers:
116
+
117
+ ```typescript
118
+ import { test, expect } from 'vitest'
119
+ import { runEvals } from '@mastra/core/evals'
120
+ import { mastra } from '../src/mastra'
121
+
122
+ test('weather workflow runs fetch-weather then plan-activities', { timeout: 60_000 }, async () => {
123
+ const result = await runEvals({
124
+ target: mastra.getWorkflow('weatherWorkflow'),
125
+ data: [
126
+ {
127
+ input: { city: 'London' },
128
+ expectedTrajectory: {
129
+ steps: [
130
+ { stepType: 'workflow_step', name: 'fetch-weather' },
131
+ { stepType: 'workflow_step', name: 'plan-activities' },
132
+ ],
133
+ },
134
+ },
135
+ {
136
+ input: { city: 'Tokyo' },
137
+ expectedTrajectory: {
138
+ steps: [
139
+ { stepType: 'workflow_step', name: 'fetch-weather' },
140
+ { stepType: 'workflow_step', name: 'plan-activities' },
141
+ ],
142
+ },
143
+ },
144
+ ],
145
+ scorers: {
146
+ trajectory: [mastra.getScorer('trajectoryAccuracyScorer')],
147
+ },
148
+ })
149
+
150
+ expect(result).toHaveScoreAbove('trajectory.code-trajectory-accuracy-scorer', 0.9)
151
+ })
152
+ ```
153
+
154
+ The `plan-activities` step in this workflow calls `mastra.getAgent('weatherAgent')` at runtime, which only works because the target was resolved with `mastra.getWorkflow()`. See the [`runEvals` reference](https://mastra.ai/reference/evals/run-evals) for the full `WorkflowScorerConfig` shape.
155
+
118
156
  ## Reading the reporter output
119
157
 
120
158
  `MastraEvalsReporter` prints a score table for every eval test after the run completes:
@@ -122,9 +160,11 @@ Available matchers:
122
160
  ```text
123
161
  Mastra Evals
124
162
 
125
- ✓ capitals agent answers with the expected city (3 items)
126
- contains-ground-truth (gate) 1.0 ✓
127
- keyword-coverage-scorer (threshold: min 0.4) 1.0 ✓
163
+ ✓ weather agent calls the tool and answers relevantly (1 item)
164
+ check-called-tool (gate) 1.0 ✓
165
+ check-no-tool-errors (gate) 1.0 ✓
166
+ check-includes (gate) 1.0 ✓
167
+ answer-relevancy-scorer (threshold: min 0.7) 0.9 ✓
128
168
 
129
169
  Eval runs: 1 (1 passed)
130
170
  ```
@@ -130,7 +130,7 @@ await observability!.deleteFeedback({
130
130
  })
131
131
  ```
132
132
 
133
- Deleted records also disappear from feedback analytics. ClickHouse uses a lightweight delete to hide rows without guaranteeing immediate physical removal, so open-source deployments must configure an [observability retention period](https://mastra.ai/reference/storage/retention) to physically purge them. Each request is marked applied once its delete succeeds; if the delete fails, the request stays unapplied, doesn't block updates to the still-visible feedback, and you retry by calling the delete API again. Open-source deployments have no background reconciler. Configure retention for every observability signal to expire deletion requests after the signal rows they protect. If any signal is unbounded, deletion requests also remain unbounded to prevent deleted data from being reintroduced. On ClickHouse, delete APIs leave the separate delta cursor table untouched. Its rows contain identifiers rather than feedback payloads and expire within two days.
133
+ Deleted records also disappear from feedback analytics. ClickHouse uses a lightweight delete to hide rows without guaranteeing immediate physical removal, so open-source deployments must configure an [observability retention period](https://mastra.ai/reference/storage/retention) to physically purge them. Each request is marked applied once its delete succeeds; if the delete fails, the request stays unapplied, doesn't block updates to the still-visible feedback, and you retry by calling the delete API again. On ClickHouse, the next review update also retries the delete and reports the feedback as not found if the retry succeeds, or saves the new status and returns the delete's error if it fails again. Open-source deployments have no background reconciler. Configure retention for every observability signal to expire deletion requests after the signal rows they protect. If any signal is unbounded, deletion requests also remain unbounded to prevent deleted data from being reintroduced. On ClickHouse, delete APIs leave the separate delta cursor table untouched. Its rows contain identifiers rather than feedback payloads and expire within two days.
134
134
 
135
135
  ## Query feedback analytics
136
136
 
@@ -93,9 +93,7 @@ Lightweight deletion is a hide-only operation that marks rows with ClickHouse's
93
93
 
94
94
  When all five observability signals have finite retention, Mastra also applies a TTL to deletion requests so they outlive the signal rows they protect. If any signal is unbounded, deletion requests remain unbounded. See [storage retention](https://mastra.ai/reference/storage/retention) for how the deletion-request TTL is calculated.
95
95
 
96
- Feedback review-status updates require `ALTER UPDATE` permission on `mastra_feedback_events` and, when delta polling is enabled, `INSERT` permission on `mastra_feedback_events_delta`. These updates wait for the ClickHouse mutation to finish on the server that receives the write, so latency depends on that server's mutation queue. Other replicas apply the mutation through the replication log, so a reader on a lagging replica can briefly see the previous status, and an inactive replica doesn't block the update. They modify the existing feedback row and preserve deletion masks: a concurrent review update cannot recreate deleted feedback. Successful updates remain available through delta polling. If a newer version of the same feedback event is ingested while a review update is in flight, the update is re-applied to that newer version; after repeated conflicts it fails with a conflict error (HTTP 409) and the caller retries.
97
-
98
- The status mutation and the delta insert are separate operations. If the delta insert fails, the API returns an error even though the status may have changed, and continued delta polling doesn't recover that notification: later polls from the same cursor never return it, and starting delta mode without a cursor subscribes at the current head. To recover, retry the review update after resolving the error, or reread the feedback with a regular list query.
96
+ Feedback review-status updates read the current row and insert a replacement, so they need `SELECT` and `INSERT` rather than `ALTER UPDATE`, and the insert materialized view publishes each update to delta polling. An update that re-runs a delete also needs the delete permissions listed in [initialization](#initialization). A review update that lands while a delete of the same feedback is in progress re-runs that delete and returns not found, so it can't recreate deleted feedback. A failed delete doesn't block review updates, but the next update retries it: if the retry succeeds, the update returns not found; if it fails again, the new status is saved but the update returns the delete's error. If a network error or a lagging replica lets a review row outlive a successful delete, the next review update of that feedback deletes it again and returns not found.
99
97
 
100
98
  ### Observability with the legacy domain
101
99
 
@@ -372,16 +370,10 @@ const observability = new ObservabilityStorageClickhouseVNext({
372
370
  await observability.init()
373
371
  ```
374
372
 
375
- In CI/CD pipelines, set `disableInit: true` on `ClickhouseStore` and run `init()` from a deployment step that uses elevated credentials. Runtime application credentials still need more than read and insert:
373
+ In CI/CD pipelines, set `disableInit: true` on `ClickhouseStore` and run `init()` from a deployment step that uses elevated credentials. Runtime application credentials need:
376
374
 
377
375
  - `SELECT` and `INSERT` on the Mastra tables.
378
376
  - `ALTER DELETE` on the observability tables you delete from. On ClickHouse 26.6 and earlier, lightweight deletes also require `ALTER UPDATE` on those tables; ClickHouse 26.7 removed that requirement.
379
- - For feedback review updates, `ALTER UPDATE(reviewStatus)` on `mastra_feedback_events` and `INSERT` on `mastra_feedback_events_delta`.
380
-
381
- ```sql
382
- GRANT ALTER UPDATE(reviewStatus) ON <database>.mastra_feedback_events TO <runtime_user>;
383
- GRANT INSERT ON <database>.mastra_feedback_events_delta TO <runtime_user>;
384
- ```
385
377
 
386
378
  ## Observability
387
379
 
@@ -861,6 +861,7 @@ Tags are exported as a JSON string in the `mastra.tags` span attribute for broad
861
861
  If traces aren't displaying or connecting as expected:
862
862
 
863
863
  - Verify OTEL SDK is initialized before Mastra (use the `--import` flag or import at the top of the entry point)
864
+ - If you see `[OtelBridge] No OpenTelemetry tracer provider is registered globally`, the bridge can't find a tracer provider. Some frameworks create a provider without registering it globally. Pass that provider directly with `new OtelBridge({ tracerProvider })`. See [setup requirements](https://mastra.ai/reference/observability/tracing/bridges/otel)
864
865
  - Ensure the `OtelBridge` is added to your observability config
865
866
  - Check that your OTEL backend is running and accessible
866
867
 
@@ -164,7 +164,7 @@ for await (const chunk of stream) {
164
164
  | `edenai/groq/openai/gpt-oss-120b` | 131K | | | | | | $0.15 | $0.60 |
165
165
  | `edenai/groq/openai/gpt-oss-20b` | 131K | | | | | | $0.07 | $0.30 |
166
166
  | `edenai/groq/openai/gpt-oss-safeguard-20b` | 131K | | | | | | $0.07 | $0.30 |
167
- | `edenai/infomaniak/mistralai/Ministral-3-14B-Instruct-2512` | 100K | | | | | | $0.34 | $0.46 |
167
+ | `edenai/infomaniak/mistralai/Ministral-3-14B-Instruct-2512` | 100K | | | | | | $0.34 | $0.45 |
168
168
  | `edenai/ionos/meta-llama/Llama-3.3-70B-Instruct` | 128K | | | | | | $0.74 | $0.74 |
169
169
  | `edenai/ionos/openai/gpt-oss-120b` | 131K | | | | | | $0.17 | $0.74 |
170
170
  | `edenai/minimax/MiniMax-M2` | 205K | | | | | | $0.30 | $1 |
@@ -265,7 +265,7 @@ for await (const chunk of stream) {
265
265
  | `edenai/qwen/qwen3.8-max` | 1.0M | | | | | | $2 | $6 |
266
266
  | `edenai/qwen/qwen3.8-max-0902` | 1.0M | | | | | | $2 | $6 |
267
267
  | `edenai/qwen/qwq-plus` | 131K | | | | | | $0.80 | $2 |
268
- | `edenai/scaleway/deepseek-v4-flash-0731` | 256K | | | | | | $0.46 | $0.91 |
268
+ | `edenai/scaleway/deepseek-v4-flash-0731` | 256K | | | | | | $0.45 | $0.91 |
269
269
  | `edenai/scaleway/gemma-3-27b-it` | 40K | | | | | | $0.29 | $0.57 |
270
270
  | `edenai/scaleway/gpt-oss-120b` | 128K | | | | | | $0.17 | $0.68 |
271
271
  | `edenai/scaleway/llama-3.3-70b-instruct` | 128K | | | | | | $1 | $1 |
@@ -43,11 +43,11 @@ for await (const chunk of stream) {
43
43
  | `kilo/~anthropic/claude-opus-latest` | 1.0M | | | | | | $4 | $20 |
44
44
  | `kilo/~anthropic/claude-sonnet-latest` | 1.0M | | | | | | $2 | $10 |
45
45
  | `kilo/~deepseek/deepseek-flash-latest` | 1.0M | | | | | | $0.04 | $1 |
46
- | `kilo/~deepseek/deepseek-pro-latest` | 1.0M | | | | | | $0.39 | $3 |
46
+ | `kilo/~deepseek/deepseek-pro-latest` | 1.0M | | | | | | $0.39 | $1 |
47
47
  | `kilo/~deepseek/deepseek-v4-flash-latest` | 1.0M | | | | | | $0.03 | $0.32 |
48
48
  | `kilo/~google/gemini-flash-latest` | 1.0M | | | | | | $0.75 | $4 |
49
49
  | `kilo/~google/gemini-pro-latest` | 1.0M | | | | | | $2 | $12 |
50
- | `kilo/~moonshotai/kimi-latest` | 1.0M | | | | | | $1 | $11 |
50
+ | `kilo/~moonshotai/kimi-latest` | 1.0M | | | | | | $1 | $12 |
51
51
  | `kilo/~openai/gpt-astra-latest` | 1.1M | | | | | | $10 | $50 |
52
52
  | `kilo/~openai/gpt-luna-latest` | 1.1M | | | | | | $0.10 | $0.50 |
53
53
  | `kilo/~openai/gpt-mini-latest` | 400K | | | | | | $0.75 | $5 |
@@ -55,7 +55,7 @@ for await (const chunk of stream) {
55
55
  | `kilo/~openai/gpt-terra-latest` | 1.1M | | | | | | $2 | $12 |
56
56
  | `kilo/~x-ai/grok-latest` | 500K | | | | | | $2 | $5 |
57
57
  | `kilo/~z-ai/glm-flash-latest` | 1.0M | | | | | | $0.04 | $0.14 |
58
- | `kilo/~z-ai/glm-latest` | 1.0M | | | | | | $0.56 | $3 |
58
+ | `kilo/~z-ai/glm-latest` | 1.0M | | | | | | $0.56 | $2 |
59
59
  | `kilo/aion-labs/aion-2.0` | 131K | | | | | | $0.80 | $2 |
60
60
  | `kilo/aion-labs/aion-3.0` | 131K | | | | | | $3 | $6 |
61
61
  | `kilo/aion-labs/aion-3.0-mini` | 131K | | | | | | $0.70 | $1 |
@@ -386,7 +386,7 @@ for await (const chunk of stream) {
386
386
  | `kilo/tencent/hy-mt2-1.8b` | 8K | | | | | | $0.04 | $0.18 |
387
387
  | `kilo/tencent/hy-mt2-30b-a3b` | 8K | | | | | | $0.07 | $0.29 |
388
388
  | `kilo/tencent/hy-mt2-7b` | 8K | | | | | | $0.07 | $0.29 |
389
- | `kilo/tencent/hy3` | 262K | | | | | | $0.13 | $0.53 |
389
+ | `kilo/tencent/hy3` | 262K | | | | | | $0.08 | $0.33 |
390
390
  | `kilo/tencent/hy3-preview` | 262K | | | | | | $0.18 | $0.60 |
391
391
  | `kilo/tencent/hy4-preview` | 1.0M | | | | | | $0.83 | $3 |
392
392
  | `kilo/thedrummer/cydonia-24b-v4.1` | 131K | | | | | | $0.30 | $0.50 |
@@ -10,17 +10,16 @@ The `runEvals` function enables batch evaluation of agents and workflows by runn
10
10
 
11
11
  ```typescript
12
12
  import { runEvals } from '@mastra/core/evals'
13
- import { myAgent } from './agents/my-agent'
14
- import { myScorer1, myScorer2 } from './scorers'
13
+ import { mastra } from './mastra'
15
14
 
16
15
  const result = await runEvals({
17
- target: myAgent,
16
+ target: mastra.getAgent('myAgent'),
18
17
  data: [
19
18
  { input: 'What is machine learning?' },
20
19
  { input: 'Explain neural networks' },
21
20
  { input: 'How does AI work?' },
22
21
  ],
23
- scorers: [myScorer1, myScorer2],
22
+ scorers: [mastra.getScorer('myScorer1'), mastra.getScorer('myScorer2')],
24
23
  targetOptions: { maxSteps: 5 },
25
24
  concurrency: 2,
26
25
  onItemComplete: ({ item, targetResult, scorerResults }) => {
@@ -33,15 +32,17 @@ console.log(`Average scores:`, result.scores)
33
32
  console.log(`Processed ${result.summary.totalItems} items`)
34
33
  ```
35
34
 
35
+ Resolve the `target` from the `Mastra` instance with `mastra.getAgent()` or `mastra.getWorkflow()` rather than importing the agent or workflow directly. A directly imported target has no Mastra instance attached, so registry lookups made during execution (a step calling `mastra.getAgent()`, a child workflow, an `.agent('agent-id')` reference, a tool calling `mastra.getWorkflow()`) fail, scores aren't persisted, and trace-based trajectory extraction is unavailable. `runEvals` logs a warning when the target isn't registered.
36
+
36
37
  ### Multi-turn evaluation
37
38
 
38
39
  ```typescript
39
40
  import { runEvals } from '@mastra/core/evals'
40
41
  import { checks } from '@mastra/evals/checks'
41
- import { weatherAgent } from './agents/weather-agent'
42
+ import { mastra } from './mastra'
42
43
 
43
44
  const result = await runEvals({
44
- target: weatherAgent,
45
+ target: mastra.getAgent('weatherAgent'),
45
46
  data: [
46
47
  {
47
48
  inputs: [
@@ -60,13 +61,16 @@ const result = await runEvals({
60
61
  ```typescript
61
62
  import { runEvals } from '@mastra/core/evals'
62
63
  import { checks } from '@mastra/evals/checks'
63
- import { faithfulnessScorer } from './scorers'
64
+ import { mastra } from './mastra'
64
65
 
65
66
  const result = await runEvals({
66
- target: myAgent,
67
+ target: mastra.getAgent('weatherAgent'),
67
68
  data: [{ input: 'What is the weather in Brooklyn?' }],
68
69
  gates: [checks.calledTool('get_weather'), checks.noToolErrors()],
69
- scorers: [{ scorer: faithfulnessScorer, threshold: 0.7 }, checks.includes('Brooklyn')],
70
+ scorers: [
71
+ { scorer: mastra.getScorer('faithfulnessScorer'), threshold: 0.7 },
72
+ checks.includes('Brooklyn'),
73
+ ],
70
74
  })
71
75
 
72
76
  result.verdict // 'passed' | 'scored' | 'failed'
@@ -173,7 +177,7 @@ import { runEvals } from '@mastra/core/evals'
173
177
  import { checks } from '@mastra/evals/checks'
174
178
 
175
179
  const result = await runEvals({
176
- target: weatherAgent,
180
+ target: mastra.getAgent('weatherAgent'),
177
181
  data: [{ input: 'What is the weather in Brooklyn?' }],
178
182
  gates: [checks.calledTool('get_weather'), checks.noToolErrors()],
179
183
  scorers: [
@@ -213,7 +217,7 @@ const myScorer = createScorer({
213
217
  })
214
218
 
215
219
  const result = await runEvals({
216
- target: chatAgent,
220
+ target: mastra.getAgent('chatAgent'),
217
221
  data: [
218
222
  {
219
223
  input: 'What is AI?',
@@ -240,7 +244,7 @@ import { createTrajectoryAccuracyScorerCode } from '@mastra/evals/scorers/code/t
240
244
  const trajectoryScorer = createTrajectoryAccuracyScorerCode()
241
245
 
242
246
  const result = await runEvals({
243
- target: chatAgent,
247
+ target: mastra.getAgent('chatAgent'),
244
248
  data: [
245
249
  {
246
250
  input: 'What is the weather in London?',
@@ -265,7 +269,7 @@ Pass execution options like `maxSteps` or `modelSettings` to customize agent beh
265
269
 
266
270
  ```typescript
267
271
  const result = await runEvals({
268
- target: chatAgent,
272
+ target: mastra.getAgent('chatAgent'),
269
273
  data: [{ input: 'Summarize this article' }, { input: 'Translate to French' }],
270
274
  scorers: [relevancyScorer],
271
275
  targetOptions: {
@@ -277,9 +281,11 @@ const result = await runEvals({
277
281
 
278
282
  ### Workflow Evaluation
279
283
 
284
+ Resolve the workflow with `mastra.getWorkflow()` so steps that call `mastra.getAgent()` or run child workflows can reach the registry:
285
+
280
286
  ```typescript
281
287
  const workflowResult = await runEvals({
282
- target: myWorkflow,
288
+ target: mastra.getWorkflow('myWorkflow'),
283
289
  data: [
284
290
  { input: { query: 'Process this data', priority: 'high' } },
285
291
  { input: { query: 'Another task', priority: 'low' } },
@@ -309,7 +315,7 @@ Add trajectory scoring to workflow evaluations to validate step execution order:
309
315
 
310
316
  ```typescript
311
317
  const workflowResult = await runEvals({
312
- target: myWorkflow,
318
+ target: mastra.getWorkflow('myWorkflow'),
313
319
  data: [
314
320
  {
315
321
  input: { query: 'Process this data' },
@@ -340,7 +346,7 @@ Use `startOptions` on individual data items to customize each workflow run. Per-
340
346
 
341
347
  ```typescript
342
348
  const result = await runEvals({
343
- target: myWorkflow,
349
+ target: mastra.getWorkflow('myWorkflow'),
344
350
  data: [
345
351
  {
346
352
  input: { query: 'hello' },
@@ -362,7 +368,7 @@ Use `inputs` to send sequential turns on a shared thread. Scorers see the accumu
362
368
 
363
369
  ```typescript
364
370
  const result = await runEvals({
365
- target: chatAgent,
371
+ target: mastra.getAgent('chatAgent'),
366
372
  data: [
367
373
  {
368
374
  inputs: ['My favorite city is Brooklyn.', 'What is the weather in my favorite city?'],
@@ -385,7 +391,7 @@ Use `turns` to attach `gates`/`scorers` to individual turns. Each per-turn asser
385
391
 
386
392
  ```typescript
387
393
  const result = await runEvals({
388
- target: chatAgent,
394
+ target: mastra.getAgent('chatAgent'),
389
395
  data: [
390
396
  {
391
397
  turns: [
@@ -221,7 +221,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
221
221
  const scorer = createTrajectoryAccuracyScorerCode()
222
222
 
223
223
  await runEvals({
224
- target: myAgent,
224
+ target: mastra.getAgent('myAgent'),
225
225
  scorers: { trajectory: [scorer] },
226
226
  data: [
227
227
  {
@@ -245,16 +245,20 @@ await runEvals({
245
245
 
246
246
  ### Evaluation modes
247
247
 
248
- The code-based scorer operates in two modes based on `strictOrder`:
248
+ The code-based scorer operates in one of three modes based on `ordering`:
249
249
 
250
- #### Strict mode (`strictOrder: true`)
250
+ #### Strict mode (`ordering: 'strict'`)
251
251
 
252
- Requires an exact match. The actual steps must match the expected steps in the same order with no extra or missing steps. Returns `1.0` for an exact match and `0.0` otherwise.
252
+ Only an exact match (the same steps in the same order, with nothing extra or missing) scores `1.0`. Anything else gets partial credit for the expected steps that matched in position, with a penalty deducted for each extra step. For example, the two expected steps followed by one extra step scores `0.75`.
253
253
 
254
- #### Relaxed mode (`strictOrder: false`, default)
254
+ #### Relaxed mode (`ordering: 'relaxed'`, default)
255
255
 
256
256
  Allows extra steps. Expected steps must appear in the correct relative order. The score is calculated based on how many expected steps were matched, with optional penalties for extra or repeated steps.
257
257
 
258
+ #### Unordered mode (`ordering: 'unordered'`)
259
+
260
+ Only checks that each expected step is present. Order is ignored, and extra steps are reported in the result but not penalized.
261
+
258
262
  ## Code-based scoring details
259
263
 
260
264
  - **Continuous scores**: Returns values between 0.0 and 1.0 in relaxed mode; binary (0 or 1) in strict mode
@@ -303,16 +307,16 @@ const scorer = createTrajectoryAccuracyScorerCode({
303
307
  { stepType: 'tool_call', name: 'fetch-tool' },
304
308
  ],
305
309
  },
306
- comparisonOptions: { strictOrder: true },
310
+ comparisonOptions: { ordering: 'strict' },
307
311
  })
308
312
 
309
313
  const result = await runEvals({
310
- target: myAgent,
314
+ target: mastra.getAgent('myAgent'),
311
315
  scorers: { trajectory: [scorer] },
312
316
  data: [{ input: 'Get my data' }],
313
317
  })
314
318
 
315
- console.log(result.scores.trajectory['trajectory-accuracy']) // 1.0
319
+ console.log(result.scores.trajectory['code-trajectory-accuracy-scorer']) // 1.0
316
320
  ```
317
321
 
318
322
  ### Agent trajectory with relaxed ordering
@@ -327,7 +331,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
327
331
  { stepType: 'tool_call', name: 'summarize-tool' },
328
332
  ],
329
333
  },
330
- comparisonOptions: { strictOrder: false },
334
+ comparisonOptions: { ordering: 'relaxed' },
331
335
  })
332
336
 
333
337
  // Agent called search-tool → log-tool → summarize-tool
@@ -354,12 +358,12 @@ const scorer = createTrajectoryAccuracyScorerCode({
354
358
  })
355
359
 
356
360
  const result = await runEvals({
357
- target: myWorkflow,
361
+ target: mastra.getWorkflow('myWorkflow'),
358
362
  scorers: { trajectory: [scorer] },
359
363
  data: [{ input: { data: 'test' } }],
360
364
  })
361
365
 
362
- console.log(result.scores.trajectory['trajectory-accuracy'])
366
+ console.log(result.scores.trajectory['code-trajectory-accuracy-scorer'])
363
367
  ```
364
368
 
365
369
  ### Comparing step data
@@ -517,7 +521,7 @@ const scorer = createTrajectoryScorerCode({
517
521
  })
518
522
 
519
523
  const result = await runEvals({
520
- target: myAgent,
524
+ target: mastra.getAgent('myAgent'),
521
525
  scorers: { trajectory: [scorer] },
522
526
  data: [
523
527
  {
@@ -583,7 +587,7 @@ const trajectoryScorer = createTrajectoryAccuracyScorerCode({
583
587
  })
584
588
 
585
589
  const result = await runEvals({
586
- target: myAgent,
590
+ target: mastra.getAgent('myAgent'),
587
591
  scorers: {
588
592
  agent: [qualityScorer], // receives raw MastraDBMessage[] output
589
593
  trajectory: [trajectoryScorer], // receives pre-extracted Trajectory
@@ -592,7 +596,7 @@ const result = await runEvals({
592
596
  })
593
597
 
594
598
  // result.scores.agent['quality'] — agent-level score
595
- // result.scores.trajectory['trajectory-accuracy'] — trajectory score
599
+ // result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
596
600
  ```
597
601
 
598
602
  ### Workflow trajectory evaluation
@@ -612,7 +616,7 @@ const workflowTrajectoryScorer = createTrajectoryAccuracyScorerCode({
612
616
  })
613
617
 
614
618
  const result = await runEvals({
615
- target: myWorkflow,
619
+ target: mastra.getWorkflow('myWorkflow'),
616
620
  scorers: {
617
621
  workflow: [outputScorer], // receives workflow output
618
622
  trajectory: [workflowTrajectoryScorer], // receives pre-extracted Trajectory from step results
@@ -621,7 +625,7 @@ const result = await runEvals({
621
625
  })
622
626
 
623
627
  // result.scores.workflow['output-quality'] — workflow-level score
624
- // result.scores.trajectory['trajectory-accuracy'] — trajectory score
628
+ // result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
625
629
  ```
626
630
 
627
631
  ## Related
@@ -186,6 +186,7 @@ const memory = new Memory({
186
186
  - Schema-less extractors are inline string extractors emitted directly in the Observer or Reflector output.
187
187
  - Dynamic extractor functions receive runtime context, including `source`, `threadId`, `resourceId`, `mainAgent`, `memory`, and `requestContext` when available.
188
188
  - `WorkingMemoryExtractor` uses the normal extractor pipeline to update working memory through the active `Memory` instance. It uses structured extraction when working memory has a JSON schema and skips OM metadata persistence, so the working memory payload isn't duplicated under OM extracted metadata.
189
+ - When `workingMemory.schema` is set, `WorkingMemoryExtractor` validates each update against that schema before saving it. If an update doesn't match, it's skipped and reported as an extraction failure while the previous working memory stays in place. Because the schema isn't sent to the model as a structured output constraint, one invalid working memory update can't fail other extractors. A `null` in an optional field is treated as not provided.
189
190
  - `observationalMemory.observation.manageWorkingMemory` adds `WorkingMemoryExtractor` and defaults `workingMemory.agentManaged` to `false`. It defaults `workingMemory.useStateSignals` to `true` when working memory is enabled.
190
191
  - Extraction failures are reported in OM marker data and don't discard other successful extracted values.
191
192
 
@@ -138,7 +138,7 @@ await mastraClient.deleteFeedback({
138
138
 
139
139
  Returns `Promise<{ success: boolean }>` from `mastraClient.deleteFeedback()` and the HTTP route. The storage domain method returns `Promise<void>`. Requests with more than 1,000 ids return `400`, and a server running `@mastra/core` older than `1.66.0` returns `501`. ClickHouse vNext, PostgreSQL vNext, DuckDB, and the in-memory store implement deletion. Every other adapter, including LibSQL and MongoDB, throws `OBSERVABILITY_STORAGE_DELETE_FEEDBACK_NOT_IMPLEMENTED`.
140
140
 
141
- On ClickHouse, deletion uses lightweight deletes on the main feedback events table to remove rows from reads, including OLAP queries, without guaranteeing immediate physical removal, so open-source deployments must configure an [observability retention period](https://mastra.ai/reference/storage/retention) to physically purge them. Each request is marked applied once its delete succeeds; if the delete fails, the request stays unapplied, doesn't block `updateFeedbackReviewStatus()` on the still-visible feedback, and you retry by calling `deleteFeedback()` again. Open-source deployments have no background reconciler. ClickHouse applies a TTL to deletion requests only when all five signals have finite retention. See [ClickHouse native TTL](https://mastra.ai/reference/storage/retention). Delete APIs leave the separate delta cursor table untouched. Its rows contain identifiers rather than feedback payloads and expire within two days.
141
+ On ClickHouse, deletion uses lightweight deletes on the main feedback events table to remove rows from reads, including OLAP queries, without guaranteeing immediate physical removal, so open-source deployments must configure an [observability retention period](https://mastra.ai/reference/storage/retention) to physically purge them. Each request is marked applied once its delete succeeds; if the delete fails, the request stays unapplied, doesn't block `updateFeedbackReviewStatus()` on the still-visible feedback, and you retry by calling `deleteFeedback()` again. `updateFeedbackReviewStatus()` also retries the delete and throws a not-found error if the retry succeeds, or saves the new status and throws the delete's error if it fails again. Open-source deployments have no background reconciler. ClickHouse applies a TTL to deletion requests only when all five signals have finite retention. See [ClickHouse native TTL](https://mastra.ai/reference/storage/retention). Delete APIs leave the separate delta cursor table untouched. Its rows contain identifiers rather than feedback payloads and expire within two days.
142
142
 
143
143
  ### `deleteScores(args)`
144
144
 
@@ -11,7 +11,7 @@ Enables bidirectional integration between Mastra tracing and OpenTelemetry infra
11
11
  ## Constructor
12
12
 
13
13
  ```typescript
14
- new OtelBridge()
14
+ new OtelBridge(config?: OtelBridgeConfig)
15
15
  ```
16
16
 
17
17
  ## Methods
@@ -118,9 +118,18 @@ const mastra = new Mastra({
118
118
  })
119
119
  ```
120
120
 
121
- ## `OpenTelemetry` setup requirements
121
+ ## Setup requirements
122
122
 
123
- The OtelBridge requires an active OpenTelemetry SDK to function. The bridge reads from OTEL's ambient context.
123
+ The bridge creates spans through an OpenTelemetry tracer provider. For Mastra spans to be exported:
124
+
125
+ - **A tracer provider must be available.** Register one globally before Mastra runs, for example by calling `sdk.start()` on `NodeSDK` from `@opentelemetry/sdk-node`, or by calling `trace.setGlobalTracerProvider(provider)`. If the provider isn't registered globally, pass it with `new OtelBridge({ tracerProvider })`.
126
+ - **A context manager must be installed for Mastra spans to nest under outer spans.** `NodeSDK` installs one automatically. Without a context manager, spans are still exported, but Mastra root spans can't read the active span, so they start new traces instead of joining the outer trace.
127
+
128
+ When no tracer provider is available, agents and workflows still run, but no Mastra spans are exported through OpenTelemetry. The bridge logs this warning once:
129
+
130
+ ```text
131
+ [OtelBridge] No OpenTelemetry tracer provider is registered globally, so Mastra spans will not be exported through OpenTelemetry. ...
132
+ ```
124
133
 
125
134
  See the [OtelBridge Guide](https://mastra.ai/integrations/observability/opentelemetry) for complete setup instructions, including how to configure OTEL instrumentation and run your application.
126
135
 
@@ -51,7 +51,7 @@ interface BaseSpan<TType extends SpanType> {
51
51
 
52
52
  /** Snapshot of the RequestContext */
53
53
  requestContext?: Record<string, any>
54
- /** Is an event span? (occurs at startTime, has no endTime) */
54
+ /** Is an event span? (point-in-time: endTime equals startTime) */
55
55
  isEvent: boolean
56
56
  }
57
57
  ```
@@ -133,7 +133,7 @@ createEventSpan<TChildType extends SpanType>(
133
133
  ): Span<TChildType>
134
134
  ```
135
135
 
136
- Creates an event span under this span. Event spans represent point-in-time occurrences with no duration.
136
+ Creates an event span under this span. Event spans represent point-in-time occurrences with no duration. They're emitted as soon as they're created, with `isEvent: true` and an `endTime` equal to their `startTime`, so stored and exported records have `endedAt` equal to `startedAt`.
137
137
 
138
138
  ## `ExportedSpan`
139
139