@mastra/evals 1.10.2 → 1.10.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/docs/SKILL.md +2 -1
- package/dist/docs/assets/SOURCE_MAP.json +1 -1
- package/dist/docs/references/docs-evals-overview.md +1 -1
- package/dist/docs/references/docs-evals-vitest-integration.md +71 -31
- package/dist/docs/references/docs-guides-build-an-eval-loop.md +395 -0
- package/dist/docs/references/reference-evals-trajectory-accuracy.md +20 -16
- package/dist/scorers/llm/faithfulness/index.d.ts.map +1 -1
- package/dist/scorers/llm/hallucination/index.d.ts.map +1 -1
- package/dist/scorers/prebuilt/index.cjs +9 -7
- package/dist/scorers/prebuilt/index.cjs.map +1 -1
- package/dist/scorers/prebuilt/index.js +9 -7
- package/dist/scorers/prebuilt/index.js.map +1 -1
- package/package.json +11 -11
|
@@ -221,7 +221,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
221
221
|
const scorer = createTrajectoryAccuracyScorerCode()
|
|
222
222
|
|
|
223
223
|
await runEvals({
|
|
224
|
-
target: myAgent,
|
|
224
|
+
target: mastra.getAgent('myAgent'),
|
|
225
225
|
scorers: { trajectory: [scorer] },
|
|
226
226
|
data: [
|
|
227
227
|
{
|
|
@@ -245,16 +245,20 @@ await runEvals({
|
|
|
245
245
|
|
|
246
246
|
### Evaluation modes
|
|
247
247
|
|
|
248
|
-
The code-based scorer operates in
|
|
248
|
+
The code-based scorer operates in one of three modes based on `ordering`:
|
|
249
249
|
|
|
250
|
-
#### Strict mode (`
|
|
250
|
+
#### Strict mode (`ordering: 'strict'`)
|
|
251
251
|
|
|
252
|
-
|
|
252
|
+
Only an exact match (the same steps in the same order, with nothing extra or missing) scores `1.0`. Anything else gets partial credit for the expected steps that matched in position, with a penalty deducted for each extra step. For example, the two expected steps followed by one extra step scores `0.75`.
|
|
253
253
|
|
|
254
|
-
#### Relaxed mode (`
|
|
254
|
+
#### Relaxed mode (`ordering: 'relaxed'`, default)
|
|
255
255
|
|
|
256
256
|
Allows extra steps. Expected steps must appear in the correct relative order. The score is calculated based on how many expected steps were matched, with optional penalties for extra or repeated steps.
|
|
257
257
|
|
|
258
|
+
#### Unordered mode (`ordering: 'unordered'`)
|
|
259
|
+
|
|
260
|
+
Only checks that each expected step is present. Order is ignored, and extra steps are reported in the result but not penalized.
|
|
261
|
+
|
|
258
262
|
## Code-based scoring details
|
|
259
263
|
|
|
260
264
|
- **Continuous scores**: Returns values between 0.0 and 1.0 in relaxed mode; binary (0 or 1) in strict mode
|
|
@@ -303,16 +307,16 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
303
307
|
{ stepType: 'tool_call', name: 'fetch-tool' },
|
|
304
308
|
],
|
|
305
309
|
},
|
|
306
|
-
comparisonOptions: {
|
|
310
|
+
comparisonOptions: { ordering: 'strict' },
|
|
307
311
|
})
|
|
308
312
|
|
|
309
313
|
const result = await runEvals({
|
|
310
|
-
target: myAgent,
|
|
314
|
+
target: mastra.getAgent('myAgent'),
|
|
311
315
|
scorers: { trajectory: [scorer] },
|
|
312
316
|
data: [{ input: 'Get my data' }],
|
|
313
317
|
})
|
|
314
318
|
|
|
315
|
-
console.log(result.scores.trajectory['trajectory-accuracy']) // 1.0
|
|
319
|
+
console.log(result.scores.trajectory['code-trajectory-accuracy-scorer']) // 1.0
|
|
316
320
|
```
|
|
317
321
|
|
|
318
322
|
### Agent trajectory with relaxed ordering
|
|
@@ -327,7 +331,7 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
327
331
|
{ stepType: 'tool_call', name: 'summarize-tool' },
|
|
328
332
|
],
|
|
329
333
|
},
|
|
330
|
-
comparisonOptions: {
|
|
334
|
+
comparisonOptions: { ordering: 'relaxed' },
|
|
331
335
|
})
|
|
332
336
|
|
|
333
337
|
// Agent called search-tool → log-tool → summarize-tool
|
|
@@ -354,12 +358,12 @@ const scorer = createTrajectoryAccuracyScorerCode({
|
|
|
354
358
|
})
|
|
355
359
|
|
|
356
360
|
const result = await runEvals({
|
|
357
|
-
target: myWorkflow,
|
|
361
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
358
362
|
scorers: { trajectory: [scorer] },
|
|
359
363
|
data: [{ input: { data: 'test' } }],
|
|
360
364
|
})
|
|
361
365
|
|
|
362
|
-
console.log(result.scores.trajectory['trajectory-accuracy'])
|
|
366
|
+
console.log(result.scores.trajectory['code-trajectory-accuracy-scorer'])
|
|
363
367
|
```
|
|
364
368
|
|
|
365
369
|
### Comparing step data
|
|
@@ -517,7 +521,7 @@ const scorer = createTrajectoryScorerCode({
|
|
|
517
521
|
})
|
|
518
522
|
|
|
519
523
|
const result = await runEvals({
|
|
520
|
-
target: myAgent,
|
|
524
|
+
target: mastra.getAgent('myAgent'),
|
|
521
525
|
scorers: { trajectory: [scorer] },
|
|
522
526
|
data: [
|
|
523
527
|
{
|
|
@@ -583,7 +587,7 @@ const trajectoryScorer = createTrajectoryAccuracyScorerCode({
|
|
|
583
587
|
})
|
|
584
588
|
|
|
585
589
|
const result = await runEvals({
|
|
586
|
-
target: myAgent,
|
|
590
|
+
target: mastra.getAgent('myAgent'),
|
|
587
591
|
scorers: {
|
|
588
592
|
agent: [qualityScorer], // receives raw MastraDBMessage[] output
|
|
589
593
|
trajectory: [trajectoryScorer], // receives pre-extracted Trajectory
|
|
@@ -592,7 +596,7 @@ const result = await runEvals({
|
|
|
592
596
|
})
|
|
593
597
|
|
|
594
598
|
// result.scores.agent['quality'] — agent-level score
|
|
595
|
-
// result.scores.trajectory['trajectory-accuracy'] — trajectory score
|
|
599
|
+
// result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
|
|
596
600
|
```
|
|
597
601
|
|
|
598
602
|
### Workflow trajectory evaluation
|
|
@@ -612,7 +616,7 @@ const workflowTrajectoryScorer = createTrajectoryAccuracyScorerCode({
|
|
|
612
616
|
})
|
|
613
617
|
|
|
614
618
|
const result = await runEvals({
|
|
615
|
-
target: myWorkflow,
|
|
619
|
+
target: mastra.getWorkflow('myWorkflow'),
|
|
616
620
|
scorers: {
|
|
617
621
|
workflow: [outputScorer], // receives workflow output
|
|
618
622
|
trajectory: [workflowTrajectoryScorer], // receives pre-extracted Trajectory from step results
|
|
@@ -621,7 +625,7 @@ const result = await runEvals({
|
|
|
621
625
|
})
|
|
622
626
|
|
|
623
627
|
// result.scores.workflow['output-quality'] — workflow-level score
|
|
624
|
-
// result.scores.trajectory['trajectory-accuracy'] — trajectory score
|
|
628
|
+
// result.scores.trajectory['code-trajectory-accuracy-scorer'] — trajectory score
|
|
625
629
|
```
|
|
626
630
|
|
|
627
631
|
## Related
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/scorers/llm/faithfulness/index.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,kBAAkB,CAAC;AAG1D,OAAO,KAAK,EAAE,yBAAyB,EAAE,0BAA0B,EAAE,MAAM,aAAa,CAAC;AAQzF,MAAM,WAAW,yBAAyB;IACxC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,OAAO,CAAC,EAAE,MAAM,EAAE,CAAC;CACpB;AAYD,wBAAgB,wBAAwB,CAAC,EACvC,KAAK,EACL,OAAO,GACR,EAAE;IACD,KAAK,EAAE,iBAAiB,CAAC;IACzB,OAAO,CAAC,EAAE,yBAAyB,CAAC;CACrC;;;;;;;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/scorers/llm/faithfulness/index.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,kBAAkB,CAAC;AAG1D,OAAO,KAAK,EAAE,yBAAyB,EAAE,0BAA0B,EAAE,MAAM,aAAa,CAAC;AAQzF,MAAM,WAAW,yBAAyB;IACxC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,OAAO,CAAC,EAAE,MAAM,EAAE,CAAC;CACpB;AAYD,wBAAgB,wBAAwB,CAAC,EACvC,KAAK,EACL,OAAO,GACR,EAAE;IACD,KAAK,EAAE,iBAAiB,CAAC;IACzB,OAAO,CAAC,EAAE,yBAAyB,CAAC;CACrC;;;;;;;6FAkEA"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/scorers/llm/hallucination/index.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,kBAAkB,CAAC;AAC1D,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAC;AAIjE,OAAO,KAAK,EAAE,yBAAyB,EAAE,0BAA0B,EAAE,MAAM,aAAa,CAAC;AAQzF,MAAM,WAAW,aAAa;IAC5B,KAAK,CAAC,EAAE,yBAAyB,CAAC;IAClC,MAAM,EAAE,0BAA0B,CAAC;IACnC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,cAAc,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IACrC,cAAc,CAAC,EAAE,cAAc,CAAC;CACjC;AAED,MAAM,WAAW,gBAAgB;IAC/B,GAAG,EAAE,aAAa,CAAC;IACnB,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IAC7B,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,IAAI,EAAE,SAAS,GAAG,gBAAgB,CAAC;CACpC;AAED,MAAM,MAAM,YAAY,GAAG,CAAC,MAAM,EAAE,gBAAgB,KAAK,MAAM,EAAE,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;AAEtF,MAAM,WAAW,0BAA0B;IACzC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,OAAO,CAAC,EAAE,MAAM,EAAE,CAAC;IACnB,UAAU,CAAC,EAAE,YAAY,CAAC;CAC3B;AAED,wBAAgB,yBAAyB,CAAC,EACxC,KAAK,EACL,OAAO,GACR,EAAE;IACD,KAAK,EAAE,iBAAiB,CAAC;IACzB,OAAO,CAAC,EAAE,0BAA0B,CAAC;CACtC;;;;;;;;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/scorers/llm/hallucination/index.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,kBAAkB,CAAC;AAC1D,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAC;AAIjE,OAAO,KAAK,EAAE,yBAAyB,EAAE,0BAA0B,EAAE,MAAM,aAAa,CAAC;AAQzF,MAAM,WAAW,aAAa;IAC5B,KAAK,CAAC,EAAE,yBAAyB,CAAC;IAClC,MAAM,EAAE,0BAA0B,CAAC;IACnC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,cAAc,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IACrC,cAAc,CAAC,EAAE,cAAc,CAAC;CACjC;AAED,MAAM,WAAW,gBAAgB;IAC/B,GAAG,EAAE,aAAa,CAAC;IACnB,OAAO,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;IAC7B,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,IAAI,EAAE,SAAS,GAAG,gBAAgB,CAAC;CACpC;AAED,MAAM,MAAM,YAAY,GAAG,CAAC,MAAM,EAAE,gBAAgB,KAAK,MAAM,EAAE,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;AAEtF,MAAM,WAAW,0BAA0B;IACzC,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,OAAO,CAAC,EAAE,MAAM,EAAE,CAAC;IACnB,UAAU,CAAC,EAAE,YAAY,CAAC;CAC3B;AAED,wBAAgB,yBAAyB,CAAC,EACxC,KAAK,EACL,OAAO,GACR,EAAE;IACD,KAAK,EAAE,iBAAiB,CAAC;IACzB,OAAO,CAAC,EAAE,0BAA0B,CAAC;CACtC;;;;;;;;6FAgFA"}
|
|
@@ -431,7 +431,7 @@ const analyzeOutputSchema$7 = {
|
|
|
431
431
|
"type": "object",
|
|
432
432
|
"properties": {
|
|
433
433
|
"groundTruthUnit": { "type": "string" },
|
|
434
|
-
"outputUnit": { "
|
|
434
|
+
"outputUnit": { "type": ["string", "null"] },
|
|
435
435
|
"matchType": {
|
|
436
436
|
"type": "string",
|
|
437
437
|
"enum": [
|
|
@@ -762,10 +762,11 @@ function createFaithfulnessScorer({ model, options }) {
|
|
|
762
762
|
});
|
|
763
763
|
}
|
|
764
764
|
}).generateScore(({ results }) => {
|
|
765
|
-
const
|
|
766
|
-
const
|
|
765
|
+
const verdicts = results.analyzeStepResult.verdicts;
|
|
766
|
+
const totalClaims = results.preprocessStepResult?.claims?.length || verdicts.length;
|
|
767
|
+
const supportedClaims = verdicts.filter((v) => v.verdict.toLowerCase().trim() === "yes").length;
|
|
767
768
|
if (totalClaims === 0) return 0;
|
|
768
|
-
return require_scorers_utils.roundToTwoDecimals(supportedClaims / totalClaims * (options?.scale || 1));
|
|
769
|
+
return require_scorers_utils.roundToTwoDecimals(Math.min(1, supportedClaims / totalClaims) * (options?.scale || 1));
|
|
769
770
|
}).generateReason({
|
|
770
771
|
description: "Reason about the results",
|
|
771
772
|
createPrompt: ({ run, results, score }) => {
|
|
@@ -1189,10 +1190,11 @@ function createHallucinationScorer({ model, options }) {
|
|
|
1189
1190
|
});
|
|
1190
1191
|
}
|
|
1191
1192
|
}).generateScore(({ results }) => {
|
|
1192
|
-
const
|
|
1193
|
-
const
|
|
1193
|
+
const verdicts = results.analyzeStepResult.verdicts;
|
|
1194
|
+
const totalStatements = results.preprocessStepResult?.claims?.length || verdicts.length;
|
|
1195
|
+
const contradictedStatements = verdicts.filter((v) => v.verdict.toLowerCase().trim() === "yes").length;
|
|
1194
1196
|
if (totalStatements === 0) return 0;
|
|
1195
|
-
return require_scorers_utils.roundToTwoDecimals(contradictedStatements / totalStatements * (options?.scale || 1));
|
|
1197
|
+
return require_scorers_utils.roundToTwoDecimals(Math.min(1, contradictedStatements / totalStatements) * (options?.scale || 1));
|
|
1196
1198
|
}).generateReason({
|
|
1197
1199
|
description: "Reason about the results",
|
|
1198
1200
|
createPrompt: async ({ run, results, score }) => {
|