ai 7.0.110 → 7.0.111
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/dist/index.d.ts +152 -41
- package/dist/index.js +232 -64
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +141 -2
- package/dist/internal/index.js +13 -1
- package/dist/internal/index.js.map +1 -1
- package/docs/03-ai-sdk-core/60-telemetry.mdx +56 -3
- package/docs/03-ai-sdk-core/65-lifecycle-callbacks.mdx +124 -2
- package/docs/07-reference/01-ai-sdk-core/14-evaluate.mdx +16 -9
- package/package.json +1 -1
- package/src/batch/batch.ts +18 -3
- package/src/evaluate/evaluate-events.ts +131 -0
- package/src/evaluate/evaluate.ts +145 -40
- package/src/evaluate/index.ts +6 -0
- package/src/evaluate/restricted-telemetry-dispatcher.ts +46 -0
- package/src/generate-text/convert-language-model-content.ts +15 -6
- package/src/generate-text/generate-text.ts +7 -2
- package/src/generate-text/resolve-generated-file-data.ts +41 -0
- package/src/generate-text/stream-language-model-call.ts +5 -4
- package/src/telemetry/create-telemetry-dispatcher.ts +12 -0
- package/src/telemetry/telemetry.ts +34 -0
- package/src/telemetry/tracing-channel.ts +2 -1
|
@@ -126,7 +126,7 @@ const result = await generateText({
|
|
|
126
126
|
});
|
|
127
127
|
```
|
|
128
128
|
|
|
129
|
-
In this example, telemetry integrations receive `runtimeContext` as `{ requestId: 'req_abc' }`. Properties set to `false` or omitted are excluded. If `telemetry.includeRuntimeContext` is omitted, no runtime context properties are included. `telemetry.includeRuntimeContext` is supported by `generateText`, `streamText`, `ToolLoopAgent`, `embed`, `embedMany`, and `
|
|
129
|
+
In this example, telemetry integrations receive `runtimeContext` as `{ requestId: 'req_abc' }`. Properties set to `false` or omitted are excluded. If `telemetry.includeRuntimeContext` is omitted, no runtime context properties are included. `telemetry.includeRuntimeContext` is supported by `generateText`, `streamText`, `ToolLoopAgent`, `embed`, `embedMany`, `rerank`, and `experimental_evaluate`.
|
|
130
130
|
|
|
131
131
|
<Note>
|
|
132
132
|
`telemetry.includeRuntimeContext` only filters telemetry integrations,
|
|
@@ -383,6 +383,29 @@ export function myIntegration(): Telemetry {
|
|
|
383
383
|
type: '(event: RerankingModelCallEndEvent) => void | PromiseLike<void>',
|
|
384
384
|
description: 'Called when an individual reranking model call completes.',
|
|
385
385
|
},
|
|
386
|
+
{
|
|
387
|
+
name: 'experimental_onEvaluateStart',
|
|
388
|
+
type: '(event: Experimental_EvaluateStartEvent) => void | PromiseLike<void>',
|
|
389
|
+
description: 'Called when an experimental evaluation operation begins.',
|
|
390
|
+
},
|
|
391
|
+
{
|
|
392
|
+
name: 'experimental_onEvaluationModelCallStart',
|
|
393
|
+
type: '(event: Experimental_EvaluationModelCallStartEvent) => void | PromiseLike<void>',
|
|
394
|
+
description:
|
|
395
|
+
'Called when the logical evaluation model call begins. The call includes any provider retries.',
|
|
396
|
+
},
|
|
397
|
+
{
|
|
398
|
+
name: 'experimental_onEvaluationModelCallEnd',
|
|
399
|
+
type: '(event: Experimental_EvaluationModelCallEndEvent) => void | PromiseLike<void>',
|
|
400
|
+
description:
|
|
401
|
+
'Called when the logical evaluation model call, including any retries, completes.',
|
|
402
|
+
},
|
|
403
|
+
{
|
|
404
|
+
name: 'experimental_onEvaluateEnd',
|
|
405
|
+
type: '(event: Experimental_EvaluateEndEvent) => void | PromiseLike<void>',
|
|
406
|
+
description:
|
|
407
|
+
'Called when an experimental evaluation operation completes.',
|
|
408
|
+
},
|
|
386
409
|
{
|
|
387
410
|
name: 'onEnd',
|
|
388
411
|
type: '(event: GenerateTextEndEvent) => void | PromiseLike<void>',
|
|
@@ -398,6 +421,9 @@ export function myIntegration(): Telemetry {
|
|
|
398
421
|
]}
|
|
399
422
|
/>
|
|
400
423
|
|
|
424
|
+
Experimental evaluation events use the four `experimental_*` callbacks above.
|
|
425
|
+
They are not included in the stable `onStart` and `onEnd` event unions.
|
|
426
|
+
|
|
401
427
|
The event types for each method are the same as the corresponding [lifecycle callbacks](/docs/ai-sdk-core/lifecycle-callbacks). See the lifecycle callbacks documentation for the full property reference of each event.
|
|
402
428
|
|
|
403
429
|
## Collected Data
|
|
@@ -524,6 +550,31 @@ For `rerank`, the integration records spans with `CLIENT` kind:
|
|
|
524
550
|
- `gen_ai.provider.name`: the provider
|
|
525
551
|
- `gen_ai.request.model`: the model ID
|
|
526
552
|
|
|
553
|
+
#### experimental_evaluate
|
|
554
|
+
|
|
555
|
+
For `experimental_evaluate`, the integration records two spans with `CLIENT`
|
|
556
|
+
kind: an **`evaluate {modelId}`** root span for the whole operation and one
|
|
557
|
+
**`evaluate {modelId}`** child span for evaluation model work, including
|
|
558
|
+
retries. Both include the provider and requested model with
|
|
559
|
+
`gen_ai.operation.name: "evaluate"`. Token
|
|
560
|
+
usage is recorded on the child span. When evaluation attributes are enabled,
|
|
561
|
+
answers are recorded on both spans so the operation span contains its output.
|
|
562
|
+
|
|
563
|
+
`evaluate` is an AI SDK-defined custom operation name, not a well-known
|
|
564
|
+
OpenTelemetry GenAI operation. OpenTelemetry permits custom operation names
|
|
565
|
+
when no well-known value describes the operation.
|
|
566
|
+
|
|
567
|
+
OpenTelemetry's `gen_ai.evaluation.result` event describes the result of
|
|
568
|
+
evaluating a GenAI output for quality or other characteristics. In contrast,
|
|
569
|
+
`experimental_evaluate` can evaluate arbitrary JSON state for classification,
|
|
570
|
+
routing, scoring, and similar tasks. The integration therefore does not emit a
|
|
571
|
+
`gen_ai.evaluation.result` event.
|
|
572
|
+
|
|
573
|
+
Evaluation state, questions, and answers also do not have applicable GenAI
|
|
574
|
+
semantic-convention attributes. Enable the `experimental_evaluation`
|
|
575
|
+
supplemental option to record them under AI SDK-specific
|
|
576
|
+
`ai.evaluation.*` attributes.
|
|
577
|
+
|
|
527
578
|
#### GenAI span details
|
|
528
579
|
|
|
529
580
|
##### GenAI message format
|
|
@@ -579,10 +630,10 @@ registerTelemetry(
|
|
|
579
630
|
|
|
580
631
|
The callback runs when each span is created and receives:
|
|
581
632
|
|
|
582
|
-
- `spanType`: the type of span being created (`operation`, `step`, `languageModel`, `tool`, `embedding`, or `
|
|
633
|
+
- `spanType`: the type of span being created (`operation`, `step`, `languageModel`, `tool`, `embedding`, `reranking`, or `experimental_evaluation`).
|
|
583
634
|
- `operationId`: the AI SDK operation ID for the current call, such as `ai.generateText` or `ai.streamText`.
|
|
584
635
|
- `callId`: the unique ID for the current AI SDK call.
|
|
585
|
-
- `runtimeContext`: the telemetry-filtered runtime context for text generation, embedding, and
|
|
636
|
+
- `runtimeContext`: the telemetry-filtered runtime context for text generation, embedding, reranking, and evaluation spans. Text generation spans also reflect updates from `prepareStep`.
|
|
586
637
|
|
|
587
638
|
Custom attributes are merged with AI SDK attributes on the span. AI SDK-owned
|
|
588
639
|
attributes take precedence when a custom attribute uses the same key.
|
|
@@ -642,6 +693,7 @@ registerTelemetry(
|
|
|
642
693
|
providerMetadata: true,
|
|
643
694
|
embedding: true,
|
|
644
695
|
reranking: true,
|
|
696
|
+
experimental_evaluation: true,
|
|
645
697
|
runtimeContext: true,
|
|
646
698
|
headers: true,
|
|
647
699
|
toolChoice: true,
|
|
@@ -656,6 +708,7 @@ The available options are:
|
|
|
656
708
|
- `providerMetadata`: `ai.response.providerMetadata` on operation, step, and model-call spans.
|
|
657
709
|
- `embedding`: embedding inputs and outputs.
|
|
658
710
|
- `reranking`: rerank input documents and ranking output.
|
|
711
|
+
- `experimental_evaluation`: experimental evaluation state, questions, and answers.
|
|
659
712
|
- `runtimeContext`: `ai.settings.context.*`.
|
|
660
713
|
- `headers`: `ai.request.headers.*`.
|
|
661
714
|
- `toolChoice`: `ai.prompt.toolChoice`.
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
---
|
|
2
2
|
title: Lifecycle Callbacks
|
|
3
|
-
description: Observe AI SDK lifecycle events in generateText, streamText, embed, embedMany, and
|
|
3
|
+
description: Observe AI SDK lifecycle events in generateText, streamText, embed, embedMany, rerank, and experimental_evaluate calls
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Lifecycle Callbacks
|
|
7
7
|
|
|
8
8
|
Event callbacks let you run your own code at important points in an AI SDK call.
|
|
9
|
-
You can attach them directly to `generateText`, `streamText`, `embed`, `embedMany`, and `
|
|
9
|
+
You can attach them directly to `generateText`, `streamText`, `embed`, `embedMany`, `rerank`, and `experimental_evaluate` calls to observe what happened, record usage, debug multi-step generations, and monitor tool execution.
|
|
10
10
|
|
|
11
11
|
They are especially useful when you want application-specific logic close to the call site:
|
|
12
12
|
|
|
@@ -449,6 +449,25 @@ const result = await generateText({
|
|
|
449
449
|
]}
|
|
450
450
|
/>
|
|
451
451
|
|
|
452
|
+
### `experimental_evaluate`
|
|
453
|
+
|
|
454
|
+
<PropertiesTable
|
|
455
|
+
content={[
|
|
456
|
+
{
|
|
457
|
+
name: 'onStart',
|
|
458
|
+
type: '(event: Experimental_EvaluateStartEvent) => void | Promise<void>',
|
|
459
|
+
description:
|
|
460
|
+
'Called when the evaluation operation begins, before the evaluation model is called.',
|
|
461
|
+
},
|
|
462
|
+
{
|
|
463
|
+
name: 'onEnd',
|
|
464
|
+
type: '(event: Experimental_EvaluateEndEvent) => void | Promise<void>',
|
|
465
|
+
description:
|
|
466
|
+
'Called when the evaluation operation completes successfully.',
|
|
467
|
+
},
|
|
468
|
+
]}
|
|
469
|
+
/>
|
|
470
|
+
|
|
452
471
|
## Event Data Reference
|
|
453
472
|
|
|
454
473
|
The exact event data depends on the callback. The tables below summarize the fields you will most commonly use.
|
|
@@ -1151,3 +1170,106 @@ For `embed`, `value` is a single string. For `embedMany`, `value` is an array of
|
|
|
1151
1170
|
},
|
|
1152
1171
|
]}
|
|
1153
1172
|
/>
|
|
1173
|
+
|
|
1174
|
+
### Evaluation Events
|
|
1175
|
+
|
|
1176
|
+
Evaluation callbacks are experimental. Their exported types are
|
|
1177
|
+
`Experimental_EvaluateStartEvent` and `Experimental_EvaluateEndEvent`.
|
|
1178
|
+
|
|
1179
|
+
#### onStart
|
|
1180
|
+
|
|
1181
|
+
<PropertiesTable
|
|
1182
|
+
content={[
|
|
1183
|
+
{
|
|
1184
|
+
name: 'callId',
|
|
1185
|
+
type: 'string',
|
|
1186
|
+
description: 'Unique identifier for this evaluation call.',
|
|
1187
|
+
},
|
|
1188
|
+
{
|
|
1189
|
+
name: 'operationId',
|
|
1190
|
+
type: "'ai.evaluate'",
|
|
1191
|
+
description: 'Evaluation operation identifier.',
|
|
1192
|
+
},
|
|
1193
|
+
{
|
|
1194
|
+
name: 'runtimeContext',
|
|
1195
|
+
type: 'RUNTIME_CONTEXT',
|
|
1196
|
+
description: 'Full user-defined runtime context.',
|
|
1197
|
+
},
|
|
1198
|
+
{
|
|
1199
|
+
name: 'provider',
|
|
1200
|
+
type: 'string',
|
|
1201
|
+
description: 'Provider identifier for the evaluation model.',
|
|
1202
|
+
},
|
|
1203
|
+
{
|
|
1204
|
+
name: 'modelId',
|
|
1205
|
+
type: 'string',
|
|
1206
|
+
description: 'Evaluation model identifier.',
|
|
1207
|
+
},
|
|
1208
|
+
{
|
|
1209
|
+
name: 'state',
|
|
1210
|
+
type: 'string | object | array',
|
|
1211
|
+
description: 'Shared state being evaluated.',
|
|
1212
|
+
},
|
|
1213
|
+
{
|
|
1214
|
+
name: 'questions',
|
|
1215
|
+
type: 'Readonly<Record<string, Experimental_EvaluationQuestion>>',
|
|
1216
|
+
description: 'Questions being evaluated against the shared state.',
|
|
1217
|
+
},
|
|
1218
|
+
{
|
|
1219
|
+
name: 'maxRetries',
|
|
1220
|
+
type: 'number',
|
|
1221
|
+
description: 'Maximum number of retries for the model call.',
|
|
1222
|
+
},
|
|
1223
|
+
{
|
|
1224
|
+
name: 'headers',
|
|
1225
|
+
type: 'Record<string, string> | undefined',
|
|
1226
|
+
description: 'Additional HTTP headers sent with the request.',
|
|
1227
|
+
},
|
|
1228
|
+
{
|
|
1229
|
+
name: 'providerOptions',
|
|
1230
|
+
type: 'ProviderOptions',
|
|
1231
|
+
description: 'Provider-specific options.',
|
|
1232
|
+
},
|
|
1233
|
+
]}
|
|
1234
|
+
/>
|
|
1235
|
+
|
|
1236
|
+
#### onEnd
|
|
1237
|
+
|
|
1238
|
+
Includes every `onStart` field plus:
|
|
1239
|
+
|
|
1240
|
+
<PropertiesTable
|
|
1241
|
+
content={[
|
|
1242
|
+
{
|
|
1243
|
+
name: 'answers',
|
|
1244
|
+
type: 'Record<string, Experimental_EvaluationAnswer>',
|
|
1245
|
+
description: 'Exactly one typed answer per question ID.',
|
|
1246
|
+
},
|
|
1247
|
+
{
|
|
1248
|
+
name: 'usage',
|
|
1249
|
+
type: '{ inputTokens: number | undefined; outputTokens: number | undefined; totalTokens: number | undefined }',
|
|
1250
|
+
description: 'Token usage for the evaluation operation.',
|
|
1251
|
+
},
|
|
1252
|
+
{
|
|
1253
|
+
name: 'warnings',
|
|
1254
|
+
type: 'Array<Warning>',
|
|
1255
|
+
description: 'Warnings from the evaluation model.',
|
|
1256
|
+
},
|
|
1257
|
+
{
|
|
1258
|
+
name: 'rounding',
|
|
1259
|
+
type: '{ probabilityDecimals?: number; scoreDecimals?: number } | undefined',
|
|
1260
|
+
description:
|
|
1261
|
+
'Provider-declared decimal precision for probabilities and scores.',
|
|
1262
|
+
},
|
|
1263
|
+
{
|
|
1264
|
+
name: 'providerMetadata',
|
|
1265
|
+
type: 'ProviderMetadata | undefined',
|
|
1266
|
+
description: 'Optional provider-specific metadata.',
|
|
1267
|
+
},
|
|
1268
|
+
{
|
|
1269
|
+
name: 'response',
|
|
1270
|
+
type: '{ id?: string; timestamp: Date; modelId: string; headers?: Record<string, string>; body?: unknown }',
|
|
1271
|
+
description:
|
|
1272
|
+
'Response metadata including the resolved model and timestamp.',
|
|
1273
|
+
},
|
|
1274
|
+
]}
|
|
1275
|
+
/>
|
|
@@ -14,15 +14,22 @@ state. See [Evaluation](/docs/ai-sdk-core/evaluation) for examples and semantics
|
|
|
14
14
|
|
|
15
15
|
## Parameters
|
|
16
16
|
|
|
17
|
-
| Parameter | Type
|
|
18
|
-
| ----------------- |
|
|
19
|
-
| `model` | `Experimental_EvaluationModel`
|
|
20
|
-
| `state` | `string \| object \| array`
|
|
21
|
-
| `questions` | `Record<string, Experimental_EvaluationQuestion>`
|
|
22
|
-
| `maxRetries` | `number`
|
|
23
|
-
| `abortSignal` | `AbortSignal`
|
|
24
|
-
| `headers` | `Record<string, string>`
|
|
25
|
-
| `providerOptions` | `ProviderOptions`
|
|
17
|
+
| Parameter | Type | Description |
|
|
18
|
+
| ----------------- | -------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
19
|
+
| `model` | `Experimental_EvaluationModel` | Required experimental v4 model instance or a string ID resolved by Gateway or an explicitly configured evaluation-capable default provider. |
|
|
20
|
+
| `state` | `string \| object \| array` | Required JSON-compatible shared state. |
|
|
21
|
+
| `questions` | `Record<string, Experimental_EvaluationQuestion>` | Required nonempty question map. |
|
|
22
|
+
| `maxRetries` | `number` | Nonnegative integer; defaults to 2. |
|
|
23
|
+
| `abortSignal` | `AbortSignal` | Cancels evaluation. |
|
|
24
|
+
| `headers` | `Record<string, string>` | Additional HTTP headers. |
|
|
25
|
+
| `providerOptions` | `ProviderOptions` | Provider-specific options. |
|
|
26
|
+
| `telemetry` | `TelemetryOptions` | Telemetry configuration, including per-call integrations, input/output recording, and a function ID. |
|
|
27
|
+
| `runtimeContext` | `Record<string, unknown>` | Context available to lifecycle callbacks and selectively included in telemetry with `telemetry.includeRuntimeContext`. |
|
|
28
|
+
| `onStart` | `(event: Experimental_EvaluateStartEvent) => void` | Called when the evaluation operation begins. |
|
|
29
|
+
| `onEnd` | `(event: Experimental_EvaluateEndEvent) => void` | Called when the evaluation operation completes successfully. |
|
|
30
|
+
|
|
31
|
+
See [Lifecycle Callbacks](/docs/ai-sdk-core/lifecycle-callbacks#experimental_evaluate)
|
|
32
|
+
for the complete `onStart` and `onEnd` event fields.
|
|
26
33
|
|
|
27
34
|
## Result
|
|
28
35
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ai",
|
|
3
|
-
"version": "7.0.
|
|
3
|
+
"version": "7.0.111",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "AI SDK by Vercel - build apps like ChatGPT, Claude, Gemini, and more with a single interface for any model using the Vercel AI Gateway or go direct to OpenAI, Anthropic, Google, or any other model provider.",
|
|
6
6
|
"license": "Apache-2.0",
|
package/src/batch/batch.ts
CHANGED
|
@@ -374,7 +374,13 @@ export function getBatchResults<TOOLS extends ToolSet>({
|
|
|
374
374
|
cancel?: (reason?: unknown) => void;
|
|
375
375
|
} = {
|
|
376
376
|
async transform(item, controller) {
|
|
377
|
-
controller.enqueue(
|
|
377
|
+
controller.enqueue(
|
|
378
|
+
await convertBatchItemResult({
|
|
379
|
+
item,
|
|
380
|
+
tools,
|
|
381
|
+
abortSignal: operationAbortSignal,
|
|
382
|
+
}),
|
|
383
|
+
);
|
|
378
384
|
},
|
|
379
385
|
|
|
380
386
|
cancel(reason) {
|
|
@@ -507,9 +513,11 @@ function validateBatchReference({
|
|
|
507
513
|
async function convertBatchItemResult<TOOLS extends ToolSet>({
|
|
508
514
|
item,
|
|
509
515
|
tools,
|
|
516
|
+
abortSignal,
|
|
510
517
|
}: {
|
|
511
518
|
item: BatchV4ItemResult;
|
|
512
519
|
tools: TOOLS | undefined;
|
|
520
|
+
abortSignal: AbortSignal | undefined;
|
|
513
521
|
}): Promise<BatchItemResult<TOOLS>> {
|
|
514
522
|
switch (item.type) {
|
|
515
523
|
case 'text':
|
|
@@ -519,7 +527,11 @@ async function convertBatchItemResult<TOOLS extends ToolSet>({
|
|
|
519
527
|
type: item.type,
|
|
520
528
|
id: item.id,
|
|
521
529
|
status: item.status,
|
|
522
|
-
...(await convertGenerateResult({
|
|
530
|
+
...(await convertGenerateResult({
|
|
531
|
+
result: item.result,
|
|
532
|
+
tools,
|
|
533
|
+
abortSignal,
|
|
534
|
+
})),
|
|
523
535
|
};
|
|
524
536
|
case 'failed':
|
|
525
537
|
return {
|
|
@@ -600,9 +612,11 @@ function convertImageResult(
|
|
|
600
612
|
async function convertGenerateResult<TOOLS extends ToolSet>({
|
|
601
613
|
result,
|
|
602
614
|
tools,
|
|
615
|
+
abortSignal,
|
|
603
616
|
}: {
|
|
604
617
|
result: LanguageModelV4GenerateResult;
|
|
605
618
|
tools: TOOLS | undefined;
|
|
619
|
+
abortSignal: AbortSignal | undefined;
|
|
606
620
|
}): Promise<TextBatchGenerationResult<TOOLS>> {
|
|
607
621
|
const toolCalls = await Promise.all(
|
|
608
622
|
result.content
|
|
@@ -620,13 +634,14 @@ async function convertGenerateResult<TOOLS extends ToolSet>({
|
|
|
620
634
|
}),
|
|
621
635
|
),
|
|
622
636
|
);
|
|
623
|
-
const content = convertLanguageModelContent<TOOLS>({
|
|
637
|
+
const content = await convertLanguageModelContent<TOOLS>({
|
|
624
638
|
content: result.content,
|
|
625
639
|
toolCalls,
|
|
626
640
|
toolOutputs: [],
|
|
627
641
|
toolApprovalRequests: [],
|
|
628
642
|
toolApprovalResponses: [],
|
|
629
643
|
tools,
|
|
644
|
+
abortSignal,
|
|
630
645
|
});
|
|
631
646
|
|
|
632
647
|
return {
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
Experimental_EvaluationModelV4CallOptions as EvaluationModelV4CallOptions,
|
|
3
|
+
Experimental_EvaluationModelV4Result as EvaluationModelV4Result,
|
|
4
|
+
} from '@ai-sdk/provider';
|
|
5
|
+
import type { Context, ProviderOptions } from '@ai-sdk/provider-utils';
|
|
6
|
+
import type { EvaluationQuestion, EvaluationResult } from './evaluation-result';
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Event passed to the `onStart` callback for evaluation operations.
|
|
10
|
+
*
|
|
11
|
+
* Called when the operation begins, before the evaluation model is called.
|
|
12
|
+
*/
|
|
13
|
+
export type EvaluateStartEvent<RUNTIME_CONTEXT extends Context = Context> = {
|
|
14
|
+
/** User-defined runtime context. */
|
|
15
|
+
readonly runtimeContext: RUNTIME_CONTEXT;
|
|
16
|
+
|
|
17
|
+
/** Unique identifier for this evaluation call, used to correlate events. */
|
|
18
|
+
readonly callId: string;
|
|
19
|
+
|
|
20
|
+
/** Identifies the operation type (`ai.evaluate`). */
|
|
21
|
+
readonly operationId: 'ai.evaluate';
|
|
22
|
+
|
|
23
|
+
/** The provider identifier. */
|
|
24
|
+
readonly provider: string;
|
|
25
|
+
|
|
26
|
+
/** The evaluation model identifier. */
|
|
27
|
+
readonly modelId: string;
|
|
28
|
+
|
|
29
|
+
/** The shared state being evaluated. */
|
|
30
|
+
readonly state: EvaluationModelV4CallOptions['state'];
|
|
31
|
+
|
|
32
|
+
/** The questions being evaluated against the shared state. */
|
|
33
|
+
readonly questions: Readonly<Record<string, EvaluationQuestion>>;
|
|
34
|
+
|
|
35
|
+
/** Maximum number of retries for the evaluation model call. */
|
|
36
|
+
readonly maxRetries: number;
|
|
37
|
+
|
|
38
|
+
/** Additional HTTP headers sent with the request. */
|
|
39
|
+
readonly headers: Record<string, string> | undefined;
|
|
40
|
+
|
|
41
|
+
/** Additional provider-specific options. */
|
|
42
|
+
readonly providerOptions: ProviderOptions;
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Event passed to the `onEnd` callback for evaluation operations.
|
|
47
|
+
*
|
|
48
|
+
* Called when the operation completes successfully.
|
|
49
|
+
*/
|
|
50
|
+
export type EvaluateEndEvent<RUNTIME_CONTEXT extends Context = Context> =
|
|
51
|
+
EvaluateStartEvent<RUNTIME_CONTEXT> & {
|
|
52
|
+
/** Exactly one typed answer per question ID. */
|
|
53
|
+
readonly answers: EvaluationResult<
|
|
54
|
+
Record<string, EvaluationQuestion>
|
|
55
|
+
>['answers'];
|
|
56
|
+
|
|
57
|
+
/** Token usage for the evaluation operation. */
|
|
58
|
+
readonly usage: EvaluationResult<
|
|
59
|
+
Record<string, EvaluationQuestion>
|
|
60
|
+
>['usage'];
|
|
61
|
+
|
|
62
|
+
/** Warnings from the evaluation model. */
|
|
63
|
+
readonly warnings: EvaluationResult<
|
|
64
|
+
Record<string, EvaluationQuestion>
|
|
65
|
+
>['warnings'];
|
|
66
|
+
|
|
67
|
+
/** Provider-declared decimal precision for probabilities and scores. */
|
|
68
|
+
readonly rounding: EvaluationResult<
|
|
69
|
+
Record<string, EvaluationQuestion>
|
|
70
|
+
>['rounding'];
|
|
71
|
+
|
|
72
|
+
/** Optional provider-specific metadata. */
|
|
73
|
+
readonly providerMetadata: EvaluationResult<
|
|
74
|
+
Record<string, EvaluationQuestion>
|
|
75
|
+
>['providerMetadata'];
|
|
76
|
+
|
|
77
|
+
/** Response metadata, including the resolved model ID and timestamp. */
|
|
78
|
+
readonly response: EvaluationResult<
|
|
79
|
+
Record<string, EvaluationQuestion>
|
|
80
|
+
>['response'];
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Event fired when the evaluation model call begins.
|
|
85
|
+
*
|
|
86
|
+
* The logical model call includes any provider retries.
|
|
87
|
+
*/
|
|
88
|
+
export type EvaluationModelCallStartEvent = {
|
|
89
|
+
/** Unique identifier for the outer evaluation call. */
|
|
90
|
+
readonly callId: string;
|
|
91
|
+
|
|
92
|
+
/** Identifies the inner operation (`ai.evaluate.doEvaluate`). */
|
|
93
|
+
readonly operationId: 'ai.evaluate.doEvaluate';
|
|
94
|
+
|
|
95
|
+
/** The provider identifier. */
|
|
96
|
+
readonly provider: string;
|
|
97
|
+
|
|
98
|
+
/** The evaluation model identifier. */
|
|
99
|
+
readonly modelId: string;
|
|
100
|
+
|
|
101
|
+
/** The shared state being evaluated. */
|
|
102
|
+
readonly state: EvaluationModelV4CallOptions['state'];
|
|
103
|
+
|
|
104
|
+
/** The questions being evaluated against the shared state. */
|
|
105
|
+
readonly questions: Readonly<Record<string, EvaluationQuestion>>;
|
|
106
|
+
};
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Event fired after the evaluation model response has been validated.
|
|
110
|
+
*
|
|
111
|
+
* Contains the result of the logical model call, including any retries.
|
|
112
|
+
*/
|
|
113
|
+
export type EvaluationModelCallEndEvent = EvaluationModelCallStartEvent & {
|
|
114
|
+
/** Exactly one answer per question ID. */
|
|
115
|
+
readonly answers: EvaluationModelV4Result['answers'];
|
|
116
|
+
|
|
117
|
+
/** Token usage reported by the evaluation model. */
|
|
118
|
+
readonly usage?: EvaluationModelV4Result['usage'];
|
|
119
|
+
|
|
120
|
+
/** Warnings from the evaluation model. */
|
|
121
|
+
readonly warnings: EvaluationModelV4Result['warnings'];
|
|
122
|
+
|
|
123
|
+
/** Provider-declared decimal precision for probabilities and scores. */
|
|
124
|
+
readonly rounding?: EvaluationModelV4Result['rounding'];
|
|
125
|
+
|
|
126
|
+
/** Optional provider-specific metadata. */
|
|
127
|
+
readonly providerMetadata?: EvaluationModelV4Result['providerMetadata'];
|
|
128
|
+
|
|
129
|
+
/** Optional raw response metadata from the provider. */
|
|
130
|
+
readonly response?: EvaluationModelV4Result['response'];
|
|
131
|
+
};
|