ai 7.0.102 → 7.0.104
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/index.d.ts +90 -6
- package/dist/index.js +900 -299
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +1 -0
- package/dist/internal/index.js +5 -1
- package/dist/internal/index.js.map +1 -1
- package/dist/test/index.d.ts +16 -2
- package/dist/test/index.js +17 -0
- package/dist/test/index.js.map +1 -1
- package/docs/03-ai-sdk-core/18-code-mode.mdx +50 -0
- package/docs/03-ai-sdk-core/19-tool-search.mdx +80 -0
- package/docs/03-ai-sdk-core/32-evaluation.mdx +255 -0
- package/docs/03-ai-sdk-core/42-batch.mdx +1 -2
- package/docs/03-ai-sdk-core/45-provider-management.mdx +10 -0
- package/docs/03-ai-sdk-core/index.mdx +6 -0
- package/docs/04-ai-sdk-ui/03-chatbot-message-persistence.mdx +35 -0
- package/docs/06-advanced/11-secure-url-fetching.mdx +8 -2
- package/docs/07-reference/01-ai-sdk-core/14-evaluate.mdx +67 -0
- package/docs/07-reference/01-ai-sdk-core/20-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/22-dynamic-tool.mdx +7 -0
- package/docs/07-reference/01-ai-sdk-core/23-tool-search.mdx +75 -0
- package/docs/07-reference/01-ai-sdk-core/32-validate-ui-messages.mdx +12 -0
- package/docs/07-reference/01-ai-sdk-core/33-safe-validate-ui-messages.mdx +12 -0
- package/docs/07-reference/01-ai-sdk-core/40-provider-registry.mdx +17 -0
- package/docs/07-reference/01-ai-sdk-core/42-custom-provider.mdx +16 -0
- package/docs/07-reference/01-ai-sdk-core/index.mdx +6 -0
- package/docs/07-reference/02-ai-sdk-ui/31-convert-to-model-messages.mdx +12 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-evaluation-unsupported-question-type-error.mdx +31 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-model-error.mdx +4 -0
- package/docs/07-reference/05-ai-sdk-errors/ai-no-such-provider-error.mdx +4 -0
- package/package.json +12 -12
- package/src/error/index.ts +1 -0
- package/src/evaluate/evaluate.ts +106 -0
- package/src/evaluate/evaluation-provider.ts +6 -0
- package/src/evaluate/evaluation-result.ts +39 -0
- package/src/evaluate/index.ts +7 -0
- package/src/evaluate/validate-evaluation.ts +298 -0
- package/src/generate-text/generate-text.ts +25 -13
- package/src/generate-text/stream-text.ts +15 -2
- package/src/generate-text/tool-caller-configuration.ts +59 -5
- package/src/global.ts +1 -0
- package/src/index.ts +2 -0
- package/src/model/resolve-model.ts +55 -10
- package/src/prompt/prepare-tools.ts +1 -1
- package/src/realtime/browser-realtime-transport.ts +12 -1
- package/src/realtime/realtime-event-channel.ts +4 -0
- package/src/realtime/realtime-session.ts +9 -4
- package/src/registry/custom-provider.ts +37 -0
- package/src/registry/index.ts +2 -0
- package/src/registry/no-such-provider-error.ts +2 -1
- package/src/registry/provider-registry.ts +65 -6
- package/src/test/evaluation-mock-model-v4.ts +27 -0
- package/src/tool-search/prepare-tool-search.ts +143 -0
- package/src/tool-search/tool-search.ts +59 -0
- package/src/ui/convert-to-model-messages.ts +3 -0
- package/src/ui/process-ui-message-stream.ts +4 -0
- package/src/ui/ui-messages.ts +5 -1
- package/src/ui/validate-ui-messages.ts +3 -0
- package/src/ui/warn-if-ui-message-has-deprecated-raw-input.ts +36 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: AI_EvaluationUnsupportedQuestionTypeError
|
|
3
|
+
description: Identify an evaluation question type that a model does not support.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# AI_EvaluationUnsupportedQuestionTypeError
|
|
7
|
+
|
|
8
|
+
This experimental error identifies a question whose type is unsupported by an
|
|
9
|
+
evaluation model. It is exported from `ai` and `@ai-sdk/provider` as
|
|
10
|
+
`Experimental_EvaluationUnsupportedQuestionTypeError`.
|
|
11
|
+
|
|
12
|
+
## Properties
|
|
13
|
+
|
|
14
|
+
- `questionId`: The ID of the unsupported question.
|
|
15
|
+
- `questionType`: The requested question type.
|
|
16
|
+
- `provider`: The provider of the evaluation model.
|
|
17
|
+
- `modelId`: The evaluation model ID.
|
|
18
|
+
- `message`: A description of the unsupported question and model. Providers can
|
|
19
|
+
supply a custom message.
|
|
20
|
+
|
|
21
|
+
## Checking for this Error
|
|
22
|
+
|
|
23
|
+
Use the marker-based `isInstance` check, which works across package copies:
|
|
24
|
+
|
|
25
|
+
```typescript
|
|
26
|
+
import { Experimental_EvaluationUnsupportedQuestionTypeError as EvaluationUnsupportedQuestionTypeError } from 'ai';
|
|
27
|
+
|
|
28
|
+
if (EvaluationUnsupportedQuestionTypeError.isInstance(error)) {
|
|
29
|
+
console.log(error.questionId, error.questionType, error.modelId);
|
|
30
|
+
}
|
|
31
|
+
```
|
|
@@ -24,3 +24,7 @@ if (NoSuchModelError.isInstance(error)) {
|
|
|
24
24
|
// Handle the error
|
|
25
25
|
}
|
|
26
26
|
```
|
|
27
|
+
|
|
28
|
+
Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
|
|
29
|
+
includes unavailable evaluation capabilities and unknown evaluation model or
|
|
30
|
+
provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
|
|
@@ -26,3 +26,7 @@ if (NoSuchProviderError.isInstance(error)) {
|
|
|
26
26
|
// Handle the error
|
|
27
27
|
}
|
|
28
28
|
```
|
|
29
|
+
|
|
30
|
+
Experimental evaluation resolution uses `modelType: 'evaluationModel'`. This
|
|
31
|
+
includes unavailable evaluation capabilities and unknown evaluation model or
|
|
32
|
+
provider IDs. See [Evaluation](/docs/ai-sdk-core/evaluation#default-provider-strings).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ai",
|
|
3
|
-
"version": "7.0.
|
|
3
|
+
"version": "7.0.104",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "AI SDK by Vercel - build apps like ChatGPT, Claude, Gemini, and more with a single interface for any model using the Vercel AI Gateway or go direct to OpenAI, Anthropic, Google, or any other model provider.",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -42,20 +42,20 @@
|
|
|
42
42
|
}
|
|
43
43
|
},
|
|
44
44
|
"dependencies": {
|
|
45
|
-
"@ai-sdk/gateway": "4.0.
|
|
46
|
-
"@ai-sdk/provider": "4.0.
|
|
47
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
45
|
+
"@ai-sdk/gateway": "4.0.84",
|
|
46
|
+
"@ai-sdk/provider": "4.0.17",
|
|
47
|
+
"@ai-sdk/provider-utils": "5.0.43"
|
|
48
48
|
},
|
|
49
49
|
"devDependencies": {
|
|
50
|
-
"@ai-sdk/amazon-bedrock": "5.0.
|
|
51
|
-
"@ai-sdk/deepseek": "3.0.
|
|
52
|
-
"@ai-sdk/google": "4.0.
|
|
53
|
-
"@ai-sdk/groq": "4.0.
|
|
54
|
-
"@ai-sdk/huggingface": "2.0.
|
|
55
|
-
"@ai-sdk/moonshotai": "3.0.
|
|
56
|
-
"@ai-sdk/openai": "4.0.
|
|
50
|
+
"@ai-sdk/amazon-bedrock": "5.0.86",
|
|
51
|
+
"@ai-sdk/deepseek": "3.0.47",
|
|
52
|
+
"@ai-sdk/google": "4.0.74",
|
|
53
|
+
"@ai-sdk/groq": "4.0.44",
|
|
54
|
+
"@ai-sdk/huggingface": "2.0.51",
|
|
55
|
+
"@ai-sdk/moonshotai": "3.0.52",
|
|
56
|
+
"@ai-sdk/openai": "4.0.69",
|
|
57
57
|
"@ai-sdk/test-server": "2.0.1",
|
|
58
|
-
"@ai-sdk/xai": "
|
|
58
|
+
"@ai-sdk/xai": "5.0.2",
|
|
59
59
|
"@edge-runtime/vm": "^5.0.0",
|
|
60
60
|
"@smithy/eventstream-codec": "^4.3.3",
|
|
61
61
|
"@smithy/util-utf8": "^4.3.3",
|
package/src/error/index.ts
CHANGED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import {
|
|
2
|
+
Experimental_EvaluationUnsupportedQuestionTypeError as EvaluationUnsupportedQuestionTypeError,
|
|
3
|
+
type Experimental_EvaluationModelV4CallOptions as EvaluationModelV4CallOptions,
|
|
4
|
+
} from '@ai-sdk/provider';
|
|
5
|
+
import {
|
|
6
|
+
withUserAgentSuffix,
|
|
7
|
+
type ProviderOptions,
|
|
8
|
+
} from '@ai-sdk/provider-utils';
|
|
9
|
+
import { resolveEvaluationModel } from '../model/resolve-model';
|
|
10
|
+
import { logWarnings } from '../logger/log-warnings';
|
|
11
|
+
import { prepareRetries } from '../util/prepare-retries';
|
|
12
|
+
import { VERSION } from '../version';
|
|
13
|
+
import type {
|
|
14
|
+
EvaluationModel,
|
|
15
|
+
EvaluationQuestion,
|
|
16
|
+
EvaluationResult,
|
|
17
|
+
} from './evaluation-result';
|
|
18
|
+
import {
|
|
19
|
+
validateEvaluationInput,
|
|
20
|
+
validateEvaluationAnswers,
|
|
21
|
+
} from './validate-evaluation';
|
|
22
|
+
|
|
23
|
+
/** Evaluate typed questions against one shared state. Experimental. */
|
|
24
|
+
export async function evaluate<
|
|
25
|
+
const QUESTIONS extends Record<string, EvaluationQuestion>,
|
|
26
|
+
>({
|
|
27
|
+
model: modelArg,
|
|
28
|
+
state,
|
|
29
|
+
questions,
|
|
30
|
+
maxRetries,
|
|
31
|
+
abortSignal,
|
|
32
|
+
headers,
|
|
33
|
+
providerOptions = {},
|
|
34
|
+
}: {
|
|
35
|
+
/** An evaluation model instance or an ID resolved by the configured default provider. */
|
|
36
|
+
model: EvaluationModel;
|
|
37
|
+
state: EvaluationModelV4CallOptions['state'];
|
|
38
|
+
questions: QUESTIONS;
|
|
39
|
+
/** Maximum retries for transient provider failures. Defaults to 2. */
|
|
40
|
+
maxRetries?: number;
|
|
41
|
+
abortSignal?: AbortSignal;
|
|
42
|
+
headers?: Record<string, string>;
|
|
43
|
+
providerOptions?: ProviderOptions;
|
|
44
|
+
}): Promise<EvaluationResult<QUESTIONS>> {
|
|
45
|
+
const model = resolveEvaluationModel(modelArg);
|
|
46
|
+
|
|
47
|
+
validateEvaluationInput({ state, questions });
|
|
48
|
+
|
|
49
|
+
for (const [questionId, question] of Object.entries(questions)) {
|
|
50
|
+
if (!model.supportedQuestionTypes.includes(question.type)) {
|
|
51
|
+
throw new EvaluationUnsupportedQuestionTypeError({
|
|
52
|
+
questionId,
|
|
53
|
+
questionType: question.type,
|
|
54
|
+
provider: model.provider,
|
|
55
|
+
modelId: model.modelId,
|
|
56
|
+
});
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const { retry } = prepareRetries({ maxRetries, abortSignal });
|
|
61
|
+
const result = await retry(() => {
|
|
62
|
+
abortSignal?.throwIfAborted();
|
|
63
|
+
return model.doEvaluate({
|
|
64
|
+
state,
|
|
65
|
+
questions,
|
|
66
|
+
abortSignal,
|
|
67
|
+
headers: withUserAgentSuffix(headers ?? {}, `ai/${VERSION}`),
|
|
68
|
+
providerOptions,
|
|
69
|
+
});
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
abortSignal?.throwIfAborted();
|
|
73
|
+
validateEvaluationAnswers({
|
|
74
|
+
questions,
|
|
75
|
+
answers: result.answers,
|
|
76
|
+
rounding: result.rounding,
|
|
77
|
+
});
|
|
78
|
+
logWarnings({
|
|
79
|
+
warnings: result.warnings,
|
|
80
|
+
provider: model.provider,
|
|
81
|
+
model: model.modelId,
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
const inputTokens = result.usage?.inputTokens;
|
|
85
|
+
const outputTokens = result.usage?.outputTokens;
|
|
86
|
+
|
|
87
|
+
return {
|
|
88
|
+
answers: result.answers as EvaluationResult<QUESTIONS>['answers'],
|
|
89
|
+
usage: {
|
|
90
|
+
inputTokens,
|
|
91
|
+
outputTokens,
|
|
92
|
+
totalTokens:
|
|
93
|
+
inputTokens != null && outputTokens != null
|
|
94
|
+
? inputTokens + outputTokens
|
|
95
|
+
: undefined,
|
|
96
|
+
},
|
|
97
|
+
warnings: result.warnings,
|
|
98
|
+
rounding: result.rounding,
|
|
99
|
+
providerMetadata: result.providerMetadata,
|
|
100
|
+
response: {
|
|
101
|
+
...result.response,
|
|
102
|
+
timestamp: result.response?.timestamp ?? new Date(),
|
|
103
|
+
modelId: result.response?.modelId ?? model.modelId,
|
|
104
|
+
},
|
|
105
|
+
};
|
|
106
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { Experimental_EvaluationModelV4 as EvaluationModelV4 } from '@ai-sdk/provider';
|
|
2
|
+
|
|
3
|
+
/** Structural extension; evaluation is not part of the stable provider contract. */
|
|
4
|
+
export type EvaluationProvider = {
|
|
5
|
+
evaluationModel?: (modelId: string) => EvaluationModelV4;
|
|
6
|
+
};
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
Experimental_EvaluationModelV4 as EvaluationModelV4,
|
|
3
|
+
Experimental_EvaluationModelV4Question as EvaluationModelV4Question,
|
|
4
|
+
Experimental_EvaluationModelV4Result as EvaluationModelV4Result,
|
|
5
|
+
} from '@ai-sdk/provider';
|
|
6
|
+
|
|
7
|
+
export type EvaluationModel = string | EvaluationModelV4;
|
|
8
|
+
export type EvaluationQuestion = EvaluationModelV4Question;
|
|
9
|
+
|
|
10
|
+
export type EvaluationAnswer<QUESTION extends EvaluationQuestion> =
|
|
11
|
+
QUESTION extends { type: 'choice'; criteria: infer CRITERIA }
|
|
12
|
+
? {
|
|
13
|
+
type: 'choice';
|
|
14
|
+
choice: Extract<keyof CRITERIA, string>;
|
|
15
|
+
probabilities?: Record<Extract<keyof CRITERIA, string>, number>;
|
|
16
|
+
}
|
|
17
|
+
: QUESTION extends { type: 'score' }
|
|
18
|
+
? { type: 'score'; score: number; probabilities?: Record<string, number> }
|
|
19
|
+
: { type: 'boolean'; probability: number };
|
|
20
|
+
|
|
21
|
+
export type EvaluationResult<
|
|
22
|
+
QUESTIONS extends Record<string, EvaluationQuestion>,
|
|
23
|
+
> = {
|
|
24
|
+
readonly answers: {
|
|
25
|
+
[ID in keyof QUESTIONS]: EvaluationAnswer<QUESTIONS[ID]>;
|
|
26
|
+
};
|
|
27
|
+
readonly usage: {
|
|
28
|
+
inputTokens: number | undefined;
|
|
29
|
+
outputTokens: number | undefined;
|
|
30
|
+
totalTokens: number | undefined;
|
|
31
|
+
};
|
|
32
|
+
readonly warnings: EvaluationModelV4Result['warnings'];
|
|
33
|
+
readonly rounding: EvaluationModelV4Result['rounding'];
|
|
34
|
+
readonly providerMetadata: EvaluationModelV4Result['providerMetadata'];
|
|
35
|
+
readonly response: NonNullable<EvaluationModelV4Result['response']> & {
|
|
36
|
+
timestamp: Date;
|
|
37
|
+
modelId: string;
|
|
38
|
+
};
|
|
39
|
+
};
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export { evaluate as experimental_evaluate } from './evaluate';
|
|
2
|
+
export type {
|
|
3
|
+
EvaluationModel as Experimental_EvaluationModel,
|
|
4
|
+
EvaluationQuestion as Experimental_EvaluationQuestion,
|
|
5
|
+
EvaluationAnswer as Experimental_EvaluationAnswer,
|
|
6
|
+
EvaluationResult as Experimental_EvaluationResult,
|
|
7
|
+
} from './evaluation-result';
|
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
import {
|
|
2
|
+
InvalidResponseDataError,
|
|
3
|
+
type Experimental_EvaluationModelV4CallOptions as EvaluationModelV4CallOptions,
|
|
4
|
+
type Experimental_EvaluationModelV4Result as EvaluationModelV4Result,
|
|
5
|
+
} from '@ai-sdk/provider';
|
|
6
|
+
import { InvalidArgumentError } from '../error/invalid-argument-error';
|
|
7
|
+
|
|
8
|
+
// Absolute tolerance for sums and means. Never renormalize provider output.
|
|
9
|
+
const tolerance = 1e-6;
|
|
10
|
+
|
|
11
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
12
|
+
if (value == null || typeof value !== 'object' || Array.isArray(value))
|
|
13
|
+
return false;
|
|
14
|
+
const prototype = Object.getPrototypeOf(value);
|
|
15
|
+
return prototype === Object.prototype || prototype === null;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
function isJSON(value: unknown, ancestors = new Set<object>()): boolean {
|
|
19
|
+
if (value === null || typeof value === 'string' || typeof value === 'boolean')
|
|
20
|
+
return true;
|
|
21
|
+
if (typeof value === 'number') return Number.isFinite(value);
|
|
22
|
+
if (typeof value !== 'object' || (!Array.isArray(value) && !isRecord(value)))
|
|
23
|
+
return false;
|
|
24
|
+
if (ancestors.has(value)) return false;
|
|
25
|
+
ancestors.add(value);
|
|
26
|
+
const valid =
|
|
27
|
+
Object.getOwnPropertySymbols(value).length === 0 &&
|
|
28
|
+
(Array.isArray(value)
|
|
29
|
+
? Array.from(value).every(item => isJSON(item, ancestors))
|
|
30
|
+
: Object.values(value).every(item => isJSON(item, ancestors)));
|
|
31
|
+
ancestors.delete(value);
|
|
32
|
+
return valid;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function isInput(value: unknown): boolean {
|
|
36
|
+
return (
|
|
37
|
+
(typeof value === 'string' || Array.isArray(value) || isRecord(value)) &&
|
|
38
|
+
isJSON(value)
|
|
39
|
+
);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function invalidInput(
|
|
43
|
+
parameter: string,
|
|
44
|
+
value: unknown,
|
|
45
|
+
message: string,
|
|
46
|
+
): never {
|
|
47
|
+
throw new InvalidArgumentError({ parameter, value, message });
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function validateEvaluationInput({
|
|
51
|
+
state,
|
|
52
|
+
questions,
|
|
53
|
+
}: EvaluationModelV4CallOptions) {
|
|
54
|
+
if (!isInput(state)) {
|
|
55
|
+
invalidInput(
|
|
56
|
+
'state',
|
|
57
|
+
state,
|
|
58
|
+
'must be a JSON-compatible string, object, or array',
|
|
59
|
+
);
|
|
60
|
+
}
|
|
61
|
+
if (!isRecord(questions) || Object.keys(questions).length === 0) {
|
|
62
|
+
invalidInput('questions', questions, 'must be a nonempty question map');
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
for (const [id, question] of Object.entries(questions)) {
|
|
66
|
+
const parameter = `questions.${id}`;
|
|
67
|
+
if (!isRecord(question) || !isInput(question.instructions)) {
|
|
68
|
+
invalidInput(
|
|
69
|
+
parameter,
|
|
70
|
+
question,
|
|
71
|
+
'instructions must be a JSON-compatible string, object, or array',
|
|
72
|
+
);
|
|
73
|
+
}
|
|
74
|
+
const criteria = question.criteria;
|
|
75
|
+
switch (question.type) {
|
|
76
|
+
case 'choice':
|
|
77
|
+
if (!isRecord(criteria) || Object.keys(criteria).length === 0) {
|
|
78
|
+
invalidInput(
|
|
79
|
+
parameter,
|
|
80
|
+
question,
|
|
81
|
+
'choice criteria must be a nonempty option map',
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
break;
|
|
85
|
+
case 'score':
|
|
86
|
+
if (!Array.isArray(criteria) || criteria.length < 2) {
|
|
87
|
+
invalidInput(
|
|
88
|
+
parameter,
|
|
89
|
+
question,
|
|
90
|
+
'score criteria must contain at least two ordered levels',
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
break;
|
|
94
|
+
case 'boolean':
|
|
95
|
+
if (criteria === undefined) continue;
|
|
96
|
+
if (
|
|
97
|
+
!isRecord(criteria) ||
|
|
98
|
+
Object.keys(criteria).some(key => key !== 'true' && key !== 'false')
|
|
99
|
+
) {
|
|
100
|
+
invalidInput(
|
|
101
|
+
parameter,
|
|
102
|
+
question,
|
|
103
|
+
'boolean criteria may only describe true and false',
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
break;
|
|
107
|
+
default:
|
|
108
|
+
invalidInput(
|
|
109
|
+
parameter,
|
|
110
|
+
question,
|
|
111
|
+
'question type must be choice, score, or boolean',
|
|
112
|
+
);
|
|
113
|
+
}
|
|
114
|
+
if (
|
|
115
|
+
!isJSON(criteria) ||
|
|
116
|
+
Object.values(criteria).some(value => value !== null && !isInput(value))
|
|
117
|
+
) {
|
|
118
|
+
invalidInput(
|
|
119
|
+
parameter,
|
|
120
|
+
question,
|
|
121
|
+
'criteria descriptions must be JSON-compatible strings, objects, arrays, or null',
|
|
122
|
+
);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function invalidAnswer(answers: unknown, message: string): never {
|
|
128
|
+
throw new InvalidResponseDataError({ data: answers, message });
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function isProbability(value: unknown): value is number {
|
|
132
|
+
return (
|
|
133
|
+
typeof value === 'number' &&
|
|
134
|
+
Number.isFinite(value) &&
|
|
135
|
+
value >= 0 &&
|
|
136
|
+
value <= 1
|
|
137
|
+
);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
function hasExactKeys(value: Record<string, unknown>, keys: string[]): boolean {
|
|
141
|
+
return (
|
|
142
|
+
Object.keys(value).length === keys.length &&
|
|
143
|
+
keys.every(key => Object.hasOwn(value, key))
|
|
144
|
+
);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
function validateDistribution(
|
|
148
|
+
value: unknown,
|
|
149
|
+
keys: string[],
|
|
150
|
+
answers: unknown,
|
|
151
|
+
id: string,
|
|
152
|
+
roundingError: number,
|
|
153
|
+
): asserts value is Record<string, number> {
|
|
154
|
+
if (
|
|
155
|
+
!isRecord(value) ||
|
|
156
|
+
!hasExactKeys(value, keys) ||
|
|
157
|
+
!Object.values(value).every(isProbability)
|
|
158
|
+
) {
|
|
159
|
+
invalidAnswer(
|
|
160
|
+
answers,
|
|
161
|
+
`Question "${id}" must have a complete distribution of finite probabilities in [0, 1].`,
|
|
162
|
+
);
|
|
163
|
+
}
|
|
164
|
+
const sum = Object.values(value).reduce<number>(
|
|
165
|
+
(total, probability) => total + (probability as number),
|
|
166
|
+
0,
|
|
167
|
+
);
|
|
168
|
+
if (Math.abs(sum - 1) > tolerance + keys.length * roundingError) {
|
|
169
|
+
invalidAnswer(
|
|
170
|
+
answers,
|
|
171
|
+
`Question "${id}" probabilities must sum to 1 within the declared rounding precision.`,
|
|
172
|
+
);
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export function validateEvaluationAnswers({
|
|
177
|
+
questions,
|
|
178
|
+
answers,
|
|
179
|
+
rounding,
|
|
180
|
+
}: {
|
|
181
|
+
questions: EvaluationModelV4CallOptions['questions'];
|
|
182
|
+
answers: unknown;
|
|
183
|
+
rounding?: EvaluationModelV4Result['rounding'];
|
|
184
|
+
}) {
|
|
185
|
+
function roundingError(decimals: number | undefined): number {
|
|
186
|
+
if (decimals === undefined) return 0;
|
|
187
|
+
if (!Number.isInteger(decimals) || decimals < 0 || decimals > 15) {
|
|
188
|
+
invalidAnswer(
|
|
189
|
+
answers,
|
|
190
|
+
'Evaluation rounding decimals must be integers between 0 and 15.',
|
|
191
|
+
);
|
|
192
|
+
}
|
|
193
|
+
return 0.5 * 10 ** -decimals;
|
|
194
|
+
}
|
|
195
|
+
const probabilityError = roundingError(rounding?.probabilityDecimals);
|
|
196
|
+
const scoreError = roundingError(rounding?.scoreDecimals);
|
|
197
|
+
if (!isRecord(answers) || !hasExactKeys(answers, Object.keys(questions))) {
|
|
198
|
+
invalidAnswer(
|
|
199
|
+
answers,
|
|
200
|
+
'Evaluation must return exactly one answer for every question.',
|
|
201
|
+
);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
for (const [id, question] of Object.entries(questions)) {
|
|
205
|
+
const answer = answers[id];
|
|
206
|
+
if (!isRecord(answer) || answer.type !== question.type) {
|
|
207
|
+
invalidAnswer(
|
|
208
|
+
answers,
|
|
209
|
+
`Question "${id}" returned an answer with the wrong type.`,
|
|
210
|
+
);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
switch (question.type) {
|
|
214
|
+
case 'choice': {
|
|
215
|
+
if (
|
|
216
|
+
typeof answer.choice !== 'string' ||
|
|
217
|
+
!Object.hasOwn(question.criteria, answer.choice)
|
|
218
|
+
) {
|
|
219
|
+
invalidAnswer(
|
|
220
|
+
answers,
|
|
221
|
+
`Question "${id}" selected an unknown option.`,
|
|
222
|
+
);
|
|
223
|
+
}
|
|
224
|
+
if (answer.probabilities !== undefined) {
|
|
225
|
+
validateDistribution(
|
|
226
|
+
answer.probabilities,
|
|
227
|
+
Object.keys(question.criteria),
|
|
228
|
+
answers,
|
|
229
|
+
id,
|
|
230
|
+
probabilityError,
|
|
231
|
+
);
|
|
232
|
+
const selected = answer.probabilities[answer.choice];
|
|
233
|
+
if (
|
|
234
|
+
Object.values(answer.probabilities).some(
|
|
235
|
+
probability => probability > selected + tolerance,
|
|
236
|
+
)
|
|
237
|
+
) {
|
|
238
|
+
invalidAnswer(
|
|
239
|
+
answers,
|
|
240
|
+
`Question "${id}" did not select a highest-probability option.`,
|
|
241
|
+
);
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
break;
|
|
245
|
+
}
|
|
246
|
+
case 'score': {
|
|
247
|
+
if (
|
|
248
|
+
typeof answer.score !== 'number' ||
|
|
249
|
+
!Number.isFinite(answer.score) ||
|
|
250
|
+
answer.score < 0 ||
|
|
251
|
+
answer.score > question.criteria.length - 1
|
|
252
|
+
) {
|
|
253
|
+
invalidAnswer(
|
|
254
|
+
answers,
|
|
255
|
+
`Question "${id}" score must be in [0, ${question.criteria.length - 1}].`,
|
|
256
|
+
);
|
|
257
|
+
}
|
|
258
|
+
if (answer.probabilities !== undefined) {
|
|
259
|
+
const keys = question.criteria.map((_, index) => String(index));
|
|
260
|
+
validateDistribution(
|
|
261
|
+
answer.probabilities,
|
|
262
|
+
keys,
|
|
263
|
+
answers,
|
|
264
|
+
id,
|
|
265
|
+
probabilityError,
|
|
266
|
+
);
|
|
267
|
+
const mean = Object.entries(answer.probabilities).reduce(
|
|
268
|
+
(total, [index, probability]) =>
|
|
269
|
+
total + Number(index) * probability,
|
|
270
|
+
0,
|
|
271
|
+
);
|
|
272
|
+
const meanRoundingError = keys.reduce(
|
|
273
|
+
(total, index) => total + Number(index) * probabilityError,
|
|
274
|
+
0,
|
|
275
|
+
);
|
|
276
|
+
if (
|
|
277
|
+
Math.abs(mean - answer.score) >
|
|
278
|
+
tolerance + meanRoundingError + scoreError
|
|
279
|
+
) {
|
|
280
|
+
invalidAnswer(
|
|
281
|
+
answers,
|
|
282
|
+
`Question "${id}" score must equal the probability-weighted mean within the declared rounding precision.`,
|
|
283
|
+
);
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
break;
|
|
287
|
+
}
|
|
288
|
+
case 'boolean':
|
|
289
|
+
if (!isProbability(answer.probability)) {
|
|
290
|
+
invalidAnswer(
|
|
291
|
+
answers,
|
|
292
|
+
`Question "${id}" must return P(true) as a finite probability in [0, 1].`,
|
|
293
|
+
);
|
|
294
|
+
}
|
|
295
|
+
break;
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
}
|
|
@@ -63,6 +63,7 @@ import type { ActiveTools } from './active-tools';
|
|
|
63
63
|
import { calculateTokensPerSecond } from './calculate-tokens-per-second';
|
|
64
64
|
import { collectToolApprovals } from './collect-tool-approvals';
|
|
65
65
|
import { convertLanguageModelContent } from './convert-language-model-content';
|
|
66
|
+
import { createToolSearchState } from '../tool-search/prepare-tool-search';
|
|
66
67
|
import { executeToolCall } from './execute-tool-call';
|
|
67
68
|
import {
|
|
68
69
|
filterActiveTools,
|
|
@@ -106,6 +107,7 @@ import type { ToolApprovalConfiguration } from './tool-approval-configuration';
|
|
|
106
107
|
import type { ToolApprovalRequestOutput } from './tool-approval-request-output';
|
|
107
108
|
import type { ToolApprovalResponseOutput } from './tool-approval-response-output';
|
|
108
109
|
import {
|
|
110
|
+
appendToolCallerMessages,
|
|
109
111
|
prepareToolsForToolCallers,
|
|
110
112
|
resolveToolCallerConfiguration,
|
|
111
113
|
type Experimental_ToolCallers,
|
|
@@ -574,6 +576,10 @@ export async function generateText<
|
|
|
574
576
|
tools,
|
|
575
577
|
toolCallers: experimental_toolCallers,
|
|
576
578
|
});
|
|
579
|
+
const prepareToolSearch = createToolSearchState({
|
|
580
|
+
tools,
|
|
581
|
+
toolCallers: resolvedToolCallers,
|
|
582
|
+
});
|
|
577
583
|
const stopConditions = asArray(stopWhen);
|
|
578
584
|
const resolvedOnStart = onStart ?? experimental_onStart;
|
|
579
585
|
const resolvedOnStepStart = onStepStart ?? experimental_onStepStart;
|
|
@@ -906,16 +912,6 @@ export async function generateText<
|
|
|
906
912
|
prepareStepResult?.system ??
|
|
907
913
|
instructionsForNextStep;
|
|
908
914
|
|
|
909
|
-
const promptMessages = await convertToLanguageModelPrompt({
|
|
910
|
-
prompt: {
|
|
911
|
-
instructions: stepInstructions,
|
|
912
|
-
messages: prepareStepResult?.messages ?? stepInputMessages,
|
|
913
|
-
},
|
|
914
|
-
supportedUrls: await stepModel.supportedUrls,
|
|
915
|
-
download,
|
|
916
|
-
provider: stepModel.provider.split('.')[0],
|
|
917
|
-
});
|
|
918
|
-
|
|
919
915
|
runtimeContext =
|
|
920
916
|
prepareStepResult?.runtimeContext ?? runtimeContext;
|
|
921
917
|
toolsContext = prepareStepResult?.toolsContext ?? toolsContext;
|
|
@@ -927,8 +923,12 @@ export async function generateText<
|
|
|
927
923
|
const {
|
|
928
924
|
executionTools: stepExecutionTools,
|
|
929
925
|
modelTools: stepModelTools,
|
|
926
|
+
toolCallerMessages,
|
|
930
927
|
} = prepareToolsForToolCallers({
|
|
931
|
-
tools: stepActiveTools,
|
|
928
|
+
tools: prepareToolSearch(stepActiveTools, {
|
|
929
|
+
toolsContext,
|
|
930
|
+
experimental_sandbox: stepSandbox,
|
|
931
|
+
}),
|
|
932
932
|
toolCallers: resolvedToolCallers,
|
|
933
933
|
});
|
|
934
934
|
const stepToolOrder = prepareStepResult?.toolOrder ?? toolOrder;
|
|
@@ -952,8 +952,20 @@ export async function generateText<
|
|
|
952
952
|
toolChoice: prepareStepResult?.toolChoice ?? toolChoice,
|
|
953
953
|
});
|
|
954
954
|
|
|
955
|
-
const stepMessages =
|
|
956
|
-
prepareStepResult?.messages ?? stepInputMessages
|
|
955
|
+
const stepMessages = appendToolCallerMessages({
|
|
956
|
+
messages: prepareStepResult?.messages ?? stepInputMessages,
|
|
957
|
+
toolCallerMessages,
|
|
958
|
+
});
|
|
959
|
+
|
|
960
|
+
const promptMessages = await convertToLanguageModelPrompt({
|
|
961
|
+
prompt: {
|
|
962
|
+
instructions: stepInstructions,
|
|
963
|
+
messages: stepMessages,
|
|
964
|
+
},
|
|
965
|
+
supportedUrls: await stepModel.supportedUrls,
|
|
966
|
+
download,
|
|
967
|
+
provider: stepModel.provider.split('.')[0],
|
|
968
|
+
});
|
|
957
969
|
|
|
958
970
|
const stepProviderOptions = mergeObjects(
|
|
959
971
|
providerOptions,
|
|
@@ -89,6 +89,7 @@ import {
|
|
|
89
89
|
executeToolsFromStream,
|
|
90
90
|
type ExecuteToolsStreamPart,
|
|
91
91
|
} from './execute-tools-from-stream';
|
|
92
|
+
import { createToolSearchState } from '../tool-search/prepare-tool-search';
|
|
92
93
|
import { executeToolCall } from './execute-tool-call';
|
|
93
94
|
import {
|
|
94
95
|
filterActiveTools,
|
|
@@ -146,6 +147,7 @@ import type {
|
|
|
146
147
|
import { toResponseMessages } from './to-response-messages';
|
|
147
148
|
import type { ToolApprovalConfiguration } from './tool-approval-configuration';
|
|
148
149
|
import {
|
|
150
|
+
appendToolCallerMessages,
|
|
149
151
|
prepareToolsForToolCallers,
|
|
150
152
|
resolveToolCallerConfiguration,
|
|
151
153
|
type Experimental_ToolCallers,
|
|
@@ -1381,6 +1383,10 @@ class DefaultStreamTextResult<
|
|
|
1381
1383
|
tools,
|
|
1382
1384
|
toolCallers: experimental_toolCallers,
|
|
1383
1385
|
});
|
|
1386
|
+
const prepareToolSearch = createToolSearchState({
|
|
1387
|
+
tools,
|
|
1388
|
+
toolCallers: resolvedToolCallers,
|
|
1389
|
+
});
|
|
1384
1390
|
|
|
1385
1391
|
const telemetryDispatcher = createRestrictedTelemetryDispatcher<
|
|
1386
1392
|
TOOLS,
|
|
@@ -2330,8 +2336,12 @@ class DefaultStreamTextResult<
|
|
|
2330
2336
|
const {
|
|
2331
2337
|
executionTools: stepExecutionTools,
|
|
2332
2338
|
modelTools: stepModelTools,
|
|
2339
|
+
toolCallerMessages,
|
|
2333
2340
|
} = prepareToolsForToolCallers({
|
|
2334
|
-
tools: stepActiveTools,
|
|
2341
|
+
tools: prepareToolSearch(stepActiveTools, {
|
|
2342
|
+
toolsContext,
|
|
2343
|
+
experimental_sandbox: stepSandbox,
|
|
2344
|
+
}),
|
|
2335
2345
|
toolCallers: resolvedToolCallers,
|
|
2336
2346
|
});
|
|
2337
2347
|
const stepToolOrder = prepareStepResult?.toolOrder ?? toolOrder;
|
|
@@ -2355,7 +2365,10 @@ class DefaultStreamTextResult<
|
|
|
2355
2365
|
toolChoice: prepareStepResult?.toolChoice ?? toolChoice,
|
|
2356
2366
|
});
|
|
2357
2367
|
|
|
2358
|
-
const stepMessages =
|
|
2368
|
+
const stepMessages = appendToolCallerMessages({
|
|
2369
|
+
messages: prepareStepResult?.messages ?? stepInputMessages,
|
|
2370
|
+
toolCallerMessages,
|
|
2371
|
+
});
|
|
2359
2372
|
currentStepMessages = stepMessages;
|
|
2360
2373
|
const stepInstructions =
|
|
2361
2374
|
prepareStepResult?.instructions ??
|