@tanstack/ai 0.26.1 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/index.d.ts +7 -6
- package/dist/esm/activities/chat/index.js +78 -27
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/mcp/manager.d.ts +25 -0
- package/dist/esm/activities/chat/mcp/manager.js +71 -0
- package/dist/esm/activities/chat/mcp/manager.js.map +1 -0
- package/dist/esm/activities/chat/mcp/types.d.ts +56 -0
- package/dist/esm/activities/chat/middleware/types.d.ts +1 -4
- package/dist/esm/activities/chat/stream/message-updaters.js +20 -8
- package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -1
- package/dist/esm/activities/chat/tools/tool-calls.d.ts +1 -1
- package/dist/esm/activities/chat/tools/tool-calls.js +2 -1
- package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
- package/dist/esm/activities/summarize/chat-stream-summarize.js +62 -3
- package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
- package/dist/esm/extend-adapter.d.ts +22 -6
- package/dist/esm/extend-adapter.js.map +1 -1
- package/dist/esm/index.d.ts +2 -0
- package/dist/esm/index.js +2 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/logger/internal-logger.d.ts +8 -0
- package/dist/esm/logger/internal-logger.js +15 -0
- package/dist/esm/logger/internal-logger.js.map +1 -1
- package/dist/esm/middlewares/otel.js +30 -6
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/types.d.ts +6 -35
- package/dist/esm/utilities/sampling-keys.d.ts +20 -0
- package/dist/esm/utilities/sampling-keys.js +20 -0
- package/dist/esm/utilities/sampling-keys.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +67 -6
- package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +6 -3
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +3 -0
- package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +10 -1
- package/skills/ai-core/adapter-configuration/references/openai-adapter.md +4 -0
- package/skills/ai-core/chat-experience/SKILL.md +95 -7
- package/skills/ai-core/middleware/SKILL.md +11 -0
- package/skills/ai-core/tool-calling/SKILL.md +287 -0
- package/src/activities/chat/index.ts +97 -35
- package/src/activities/chat/mcp/manager.ts +85 -0
- package/src/activities/chat/mcp/types.ts +66 -0
- package/src/activities/chat/middleware/types.ts +1 -4
- package/src/activities/chat/stream/message-updaters.ts +22 -9
- package/src/activities/chat/tools/tool-calls.ts +2 -0
- package/src/activities/summarize/chat-stream-summarize.ts +162 -3
- package/src/extend-adapter.ts +42 -24
- package/src/index.ts +10 -0
- package/src/logger/internal-logger.ts +18 -0
- package/src/middlewares/otel.ts +48 -6
- package/src/types.ts +6 -35
- package/src/utilities/sampling-keys.ts +28 -0
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import type { ServerTool } from '../tools/tool-definition'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Minimal structural shape that `chat({ mcp })` needs from an MCP client.
|
|
5
|
+
*
|
|
6
|
+
* `@tanstack/ai-mcp`'s `MCPClient` and `MCPClients` satisfy this interface by
|
|
7
|
+
* shape — the core `@tanstack/ai` package does NOT import `@tanstack/ai-mcp`
|
|
8
|
+
* (ai-mcp depends on ai, not the reverse).
|
|
9
|
+
*/
|
|
10
|
+
export interface MCPToolSource {
|
|
11
|
+
// Keep the options shape in sync with ai-mcp's `ToolsOptions` — extra
|
|
12
|
+
// optional fields added there still match structurally, but chat() only
|
|
13
|
+
// forwards what is declared here.
|
|
14
|
+
tools: (options?: { lazy?: boolean }) => Promise<Array<ServerTool>>
|
|
15
|
+
close: () => Promise<void>
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Controls what happens to MCP connections when the chat run ends.
|
|
20
|
+
*
|
|
21
|
+
* - `'close'` (default) — `chat()` closes each connection when the run ends
|
|
22
|
+
* (after the agent loop completes and the stream is drained), so tools can
|
|
23
|
+
* still execute throughout the run.
|
|
24
|
+
* - `'keep-alive'` — `chat()` never closes the connections; the caller owns
|
|
25
|
+
* their lifecycle (e.g. keep them warm across requests).
|
|
26
|
+
*/
|
|
27
|
+
export type MCPConnectionPolicy = 'close' | 'keep-alive'
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Options controlling MCP tool discovery and lifecycle for a `chat()` call.
|
|
31
|
+
*/
|
|
32
|
+
export interface ChatMCPOptions {
|
|
33
|
+
/**
|
|
34
|
+
* The MCP clients or client pools to discover tools from and manage.
|
|
35
|
+
*/
|
|
36
|
+
clients: Array<MCPToolSource>
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Connection lifecycle policy applied to all clients when the run ends.
|
|
40
|
+
*
|
|
41
|
+
* Defaults to `'close'`.
|
|
42
|
+
*/
|
|
43
|
+
connection?: MCPConnectionPolicy
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* When `true`, tool schemas are fetched lazily (forwarded to
|
|
47
|
+
* `tools({ lazy: true })`).
|
|
48
|
+
*
|
|
49
|
+
* Defaults to `false`.
|
|
50
|
+
*/
|
|
51
|
+
lazyTools?: boolean
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Called when tool discovery fails for a single source.
|
|
55
|
+
*
|
|
56
|
+
* - Throw (or re-throw) from this handler to fail the entire chat call fast.
|
|
57
|
+
* - Return normally to skip that source and continue with remaining clients.
|
|
58
|
+
* - Omit this handler entirely to rethrow the error (fail-fast by default).
|
|
59
|
+
*
|
|
60
|
+
* Async handlers are awaited, so a rejected promise also fails fast.
|
|
61
|
+
*/
|
|
62
|
+
onDiscoveryError?: (
|
|
63
|
+
error: unknown,
|
|
64
|
+
source: MCPToolSource,
|
|
65
|
+
) => void | Promise<void>
|
|
66
|
+
}
|
|
@@ -90,7 +90,7 @@ export interface ChatMiddlewareContext<TContext = unknown> {
|
|
|
90
90
|
systemPrompts: Array<SystemPrompt>
|
|
91
91
|
/** Names of configured tools, if any */
|
|
92
92
|
toolNames?: Array<string>
|
|
93
|
-
/** Flattened generation options (
|
|
93
|
+
/** Flattened generation options (metadata) */
|
|
94
94
|
options?: Record<string, unknown> | undefined
|
|
95
95
|
/** Provider-specific model options */
|
|
96
96
|
modelOptions?: Record<string, unknown> | undefined
|
|
@@ -130,9 +130,6 @@ export interface ChatMiddlewareConfig {
|
|
|
130
130
|
messages: Array<ModelMessage>
|
|
131
131
|
systemPrompts: Array<SystemPrompt>
|
|
132
132
|
tools: Array<Tool>
|
|
133
|
-
temperature?: number
|
|
134
|
-
topP?: number
|
|
135
|
-
maxTokens?: number
|
|
136
133
|
metadata?: Record<string, unknown> | undefined
|
|
137
134
|
modelOptions?: Record<string, unknown> | undefined
|
|
138
135
|
}
|
|
@@ -161,10 +161,14 @@ export function updateToolCallApproval(
|
|
|
161
161
|
)
|
|
162
162
|
|
|
163
163
|
if (toolCallPart) {
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
164
|
+
const index = parts.indexOf(toolCallPart)
|
|
165
|
+
parts[index] = {
|
|
166
|
+
...toolCallPart,
|
|
167
|
+
state: 'approval-requested',
|
|
168
|
+
approval: {
|
|
169
|
+
id: approvalId,
|
|
170
|
+
needsApproval: true,
|
|
171
|
+
},
|
|
168
172
|
}
|
|
169
173
|
}
|
|
170
174
|
|
|
@@ -192,7 +196,8 @@ export function updateToolCallState(
|
|
|
192
196
|
)
|
|
193
197
|
|
|
194
198
|
if (toolCallPart) {
|
|
195
|
-
|
|
199
|
+
const index = parts.indexOf(toolCallPart)
|
|
200
|
+
parts[index] = { ...toolCallPart, state }
|
|
196
201
|
}
|
|
197
202
|
|
|
198
203
|
return { ...msg, parts }
|
|
@@ -217,8 +222,12 @@ export function updateToolCallWithOutput(
|
|
|
217
222
|
)
|
|
218
223
|
|
|
219
224
|
if (toolCallPart) {
|
|
220
|
-
|
|
221
|
-
|
|
225
|
+
const index = parts.indexOf(toolCallPart)
|
|
226
|
+
parts[index] = {
|
|
227
|
+
...toolCallPart,
|
|
228
|
+
output: errorText ? { error: errorText } : output,
|
|
229
|
+
state: state ?? (errorText ? 'input-complete' : 'complete'),
|
|
230
|
+
}
|
|
222
231
|
}
|
|
223
232
|
|
|
224
233
|
return { ...msg, parts }
|
|
@@ -242,8 +251,12 @@ export function updateToolCallApprovalResponse(
|
|
|
242
251
|
)
|
|
243
252
|
|
|
244
253
|
if (toolCallPart && toolCallPart.approval) {
|
|
245
|
-
|
|
246
|
-
|
|
254
|
+
const index = parts.indexOf(toolCallPart)
|
|
255
|
+
parts[index] = {
|
|
256
|
+
...toolCallPart,
|
|
257
|
+
approval: { ...toolCallPart.approval, approved },
|
|
258
|
+
state: 'approval-responded',
|
|
259
|
+
}
|
|
247
260
|
}
|
|
248
261
|
|
|
249
262
|
return { ...msg, parts }
|
|
@@ -599,6 +599,7 @@ export async function* executeToolCalls<TContext = unknown>(
|
|
|
599
599
|
) => CustomEvent,
|
|
600
600
|
middlewareHooks?: ToolExecutionMiddlewareHooks,
|
|
601
601
|
userContext?: TContext,
|
|
602
|
+
abortSignal?: AbortSignal,
|
|
602
603
|
): AsyncGenerator<CustomEvent, ExecuteToolCallsResult, void> {
|
|
603
604
|
const results: Array<ToolResult> = []
|
|
604
605
|
const needsApproval: Array<ApprovalRequest> = []
|
|
@@ -679,6 +680,7 @@ export async function* executeToolCalls<TContext = unknown>(
|
|
|
679
680
|
const context = {
|
|
680
681
|
toolCallId: toolCall.id,
|
|
681
682
|
context: userContext,
|
|
683
|
+
abortSignal,
|
|
682
684
|
emitCustomEvent: (eventName: string, value: Record<string, any>) => {
|
|
683
685
|
if (createCustomEventChunk) {
|
|
684
686
|
pendingEvents.push(
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { EventType } from '@ag-ui/core'
|
|
2
2
|
import { toRunErrorPayload } from '../error-payload'
|
|
3
|
+
import { MAX_TOKENS_KEYS } from '../../utilities/sampling-keys'
|
|
3
4
|
import { BaseSummarizeAdapter } from './adapter'
|
|
4
5
|
import type {
|
|
5
6
|
StreamChunk,
|
|
@@ -23,6 +24,139 @@ export interface ChatStreamCapable {
|
|
|
23
24
|
chatStream: (options: TextOptions<any>) => AsyncIterable<StreamChunk>
|
|
24
25
|
}
|
|
25
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Provider-native max-output-tokens key per summarize-adapter `name`. summarize
|
|
29
|
+
* is provider-agnostic and forwards `modelOptions` opaquely to the wrapped text
|
|
30
|
+
* adapter, so `maxLength` must be written under the exact key the underlying
|
|
31
|
+
* provider reads — no adapter reads a generic `maxTokens`. Ollama is the one
|
|
32
|
+
* exception: it nests sampling under `options`, so it has no entry here and is
|
|
33
|
+
* handled as a special nested case in `applyMaxLength`/`applyDefaultTemperature`.
|
|
34
|
+
*
|
|
35
|
+
* Keep in sync with each adapter's wire mapping:
|
|
36
|
+
* - OpenAI (Responses): `max_output_tokens`
|
|
37
|
+
* - Anthropic / Grok: `max_tokens`
|
|
38
|
+
* - Groq: `max_completion_tokens`
|
|
39
|
+
* - Gemini: `maxOutputTokens`
|
|
40
|
+
* - OpenRouter: `maxCompletionTokens`
|
|
41
|
+
* - Ollama: nested `options.num_predict` (no entry — see `applyMaxLength`)
|
|
42
|
+
*/
|
|
43
|
+
const MAX_TOKENS_KEY_BY_ADAPTER: Record<string, string> = {
|
|
44
|
+
openai: 'max_output_tokens',
|
|
45
|
+
anthropic: 'max_tokens',
|
|
46
|
+
grok: 'max_tokens',
|
|
47
|
+
groq: 'max_completion_tokens',
|
|
48
|
+
gemini: 'maxOutputTokens',
|
|
49
|
+
openrouter: 'maxCompletionTokens',
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Every flat key any supported provider uses to cap output tokens (plus the
|
|
54
|
+
* generic `maxTokens` spelling no adapter reads). Used to detect a
|
|
55
|
+
* caller-supplied token limit so the summarize default never overrides an
|
|
56
|
+
* explicit caller value. Shared with the OTel middleware via
|
|
57
|
+
* `MAX_TOKENS_KEYS` so the two spelling sets cannot drift.
|
|
58
|
+
*/
|
|
59
|
+
const KNOWN_MAX_TOKENS_KEYS = MAX_TOKENS_KEYS
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Whether `applyMaxLength` knows how to place a token limit for this adapter
|
|
63
|
+
* `name` (either the nested Ollama shape or a flat provider-native key).
|
|
64
|
+
* Used to surface a warning when `maxLength` would otherwise be silently
|
|
65
|
+
* dropped for an unrecognised adapter name.
|
|
66
|
+
*/
|
|
67
|
+
function isKnownMaxTokensAdapter(adapterName: string): boolean {
|
|
68
|
+
return (
|
|
69
|
+
adapterName === 'ollama' ||
|
|
70
|
+
MAX_TOKENS_KEY_BY_ADAPTER[adapterName] !== undefined
|
|
71
|
+
)
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Apply the low-temperature summarize default to a working copy of the
|
|
76
|
+
* caller's `modelOptions`, placed where the wrapped provider actually reads
|
|
77
|
+
* it (nested under `options` for Ollama, flat otherwise). The caller always
|
|
78
|
+
* wins: if they already set `temperature` in that location, it is untouched.
|
|
79
|
+
*/
|
|
80
|
+
function applyDefaultTemperature(
|
|
81
|
+
adapterName: string,
|
|
82
|
+
temperature: number,
|
|
83
|
+
modelOptions: Record<string, unknown>,
|
|
84
|
+
): Record<string, unknown> {
|
|
85
|
+
const merged: Record<string, unknown> = { ...modelOptions }
|
|
86
|
+
|
|
87
|
+
if (adapterName === 'ollama') {
|
|
88
|
+
const existing =
|
|
89
|
+
merged.options && typeof merged.options === 'object'
|
|
90
|
+
? (merged.options as Record<string, unknown>)
|
|
91
|
+
: undefined
|
|
92
|
+
if (existing && 'temperature' in existing) return merged
|
|
93
|
+
merged.options = { temperature, ...existing }
|
|
94
|
+
return merged
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
if ('temperature' in merged) return merged
|
|
98
|
+
merged.temperature = temperature
|
|
99
|
+
return merged
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Resolve `maxLength` to the provider-native max-output-tokens key for the
|
|
104
|
+
* given summarize-adapter `name` (this wrapper's OWN `name`, not the wrapped
|
|
105
|
+
* text adapter's) and merge it into a working copy of the caller's
|
|
106
|
+
* `modelOptions`. The caller always wins: if they already set any recognised
|
|
107
|
+
* token-limit key (flat or, for Ollama, nested `options.num_predict`), the
|
|
108
|
+
* default is left untouched. Unknown/unrecognised adapter names fall back to
|
|
109
|
+
* NOT setting a token key (the prompt hint still asks the model to stay under
|
|
110
|
+
* `maxLength`) rather than writing a dead key no provider reads.
|
|
111
|
+
*
|
|
112
|
+
* Caveat (intentional): "caller wins" keys off ANY recognised spelling in
|
|
113
|
+
* `KNOWN_MAX_TOKENS_KEYS`, but only the adapter's native key is read on the
|
|
114
|
+
* wire. So a caller who sets a NON-native spelling for this provider — e.g.
|
|
115
|
+
* `maxTokens`, or Anthropic's `max_tokens` against an OpenAI adapter — suppresses
|
|
116
|
+
* the summarize default WITHOUT getting their own value applied either: neither
|
|
117
|
+
* cap reaches the wire. This favours never clobbering a migration leftover over
|
|
118
|
+
* guaranteeing a cap; the prompt-level hint still asks the model to stay under
|
|
119
|
+
* `maxLength`. Rename the key to the provider-native spelling to forward it.
|
|
120
|
+
*/
|
|
121
|
+
function applyMaxLength(
|
|
122
|
+
adapterName: string,
|
|
123
|
+
maxLength: number,
|
|
124
|
+
modelOptions: Record<string, unknown>,
|
|
125
|
+
): Record<string, unknown> {
|
|
126
|
+
const merged: Record<string, unknown> = { ...modelOptions }
|
|
127
|
+
|
|
128
|
+
if (adapterName === 'ollama') {
|
|
129
|
+
// Honor a caller-set limit in either shape: a recognised flat key (e.g.
|
|
130
|
+
// left over from a migration) or the nested `options.num_predict`.
|
|
131
|
+
const callerSetFlatLimit = KNOWN_MAX_TOKENS_KEYS.some(
|
|
132
|
+
(k) => typeof merged[k] === 'number',
|
|
133
|
+
)
|
|
134
|
+
const existing =
|
|
135
|
+
merged.options && typeof merged.options === 'object'
|
|
136
|
+
? (merged.options as Record<string, unknown>)
|
|
137
|
+
: undefined
|
|
138
|
+
if (
|
|
139
|
+
callerSetFlatLimit ||
|
|
140
|
+
(existing && typeof existing.num_predict === 'number')
|
|
141
|
+
) {
|
|
142
|
+
return merged
|
|
143
|
+
}
|
|
144
|
+
merged.options = { num_predict: maxLength, ...existing }
|
|
145
|
+
return merged
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const key = MAX_TOKENS_KEY_BY_ADAPTER[adapterName]
|
|
149
|
+
if (key === undefined) return merged
|
|
150
|
+
|
|
151
|
+
const callerSetLimit = KNOWN_MAX_TOKENS_KEYS.some(
|
|
152
|
+
(k) => typeof merged[k] === 'number',
|
|
153
|
+
)
|
|
154
|
+
if (callerSetLimit) return merged
|
|
155
|
+
|
|
156
|
+
merged[key] = maxLength
|
|
157
|
+
return merged
|
|
158
|
+
}
|
|
159
|
+
|
|
26
160
|
/**
|
|
27
161
|
* Extract the per-model `modelOptions` type a text adapter accepts. Used by
|
|
28
162
|
* provider summarize factories so their `modelOptions` IntelliSense matches
|
|
@@ -195,13 +329,38 @@ export class ChatStreamSummarizeAdapter<
|
|
|
195
329
|
options: SummarizationOptions<TProviderOptions>,
|
|
196
330
|
systemPrompt: string,
|
|
197
331
|
): TextOptions<TProviderOptions> {
|
|
332
|
+
// Sampling knobs now live in provider-native `modelOptions`. Apply the
|
|
333
|
+
// low-temperature default where the wrapped provider actually reads it
|
|
334
|
+
// (nested under `options` for Ollama, flat otherwise) so callers can still
|
|
335
|
+
// override it. Resolving the placement from this summarize adapter's OWN
|
|
336
|
+
// `name` keeps the default off the wire correctly per provider — a flat
|
|
337
|
+
// `temperature` would be silently dropped by Ollama while still showing up
|
|
338
|
+
// in OTel.
|
|
339
|
+
let working: Record<string, unknown> = {
|
|
340
|
+
...(options.modelOptions as Record<string, unknown> | undefined),
|
|
341
|
+
}
|
|
342
|
+
working = applyDefaultTemperature(this.name, 0.3, working)
|
|
343
|
+
// `maxLength` must reach the wire under the provider-native token key (it
|
|
344
|
+
// differs per provider, and no adapter reads a generic `maxTokens`).
|
|
345
|
+
// Resolve it from this summarize adapter's `name` (the constructor arg,
|
|
346
|
+
// not the wrapped text adapter's name), never overriding a caller-supplied
|
|
347
|
+
// token limit.
|
|
348
|
+
if (options.maxLength !== undefined) {
|
|
349
|
+
if (!isKnownMaxTokensAdapter(this.name)) {
|
|
350
|
+
options.logger.warn(
|
|
351
|
+
`summarize: maxLength=${options.maxLength} could not be mapped to a provider token key for adapter name "${this.name}" — it was dropped from modelOptions (the prompt still asks the model to stay under it). Construct ChatStreamSummarizeAdapter with a recognised provider name to forward the cap.`,
|
|
352
|
+
{ provider: this.name },
|
|
353
|
+
)
|
|
354
|
+
}
|
|
355
|
+
working = applyMaxLength(this.name, options.maxLength, working)
|
|
356
|
+
}
|
|
357
|
+
const modelOptions = working as TProviderOptions
|
|
358
|
+
|
|
198
359
|
return {
|
|
199
360
|
model: options.model,
|
|
200
361
|
messages: [{ role: 'user', content: options.text }],
|
|
201
362
|
systemPrompts: [systemPrompt],
|
|
202
|
-
|
|
203
|
-
temperature: 0.3,
|
|
204
|
-
modelOptions: options.modelOptions,
|
|
363
|
+
modelOptions,
|
|
205
364
|
logger: options.logger,
|
|
206
365
|
}
|
|
207
366
|
}
|
package/src/extend-adapter.ts
CHANGED
|
@@ -144,6 +144,14 @@ type ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =
|
|
|
144
144
|
// Factory Type Inference
|
|
145
145
|
// ===========================
|
|
146
146
|
|
|
147
|
+
/**
|
|
148
|
+
* The widest factory shape `extendAdapter` accepts: any function taking a
|
|
149
|
+
* model as its first parameter. Parameters are contravariant, so `never`
|
|
150
|
+
* params and an `unknown` return accept every factory without resorting
|
|
151
|
+
* to `any`.
|
|
152
|
+
*/
|
|
153
|
+
type AnyAdapterFactory = (model: never, ...args: Array<never>) => unknown
|
|
154
|
+
|
|
147
155
|
/**
|
|
148
156
|
* Infer the model parameter type from an adapter factory function.
|
|
149
157
|
* For generic functions like `<T extends Union>(model: T)`, this gets `T` which
|
|
@@ -151,32 +159,44 @@ type ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =
|
|
|
151
159
|
*/
|
|
152
160
|
type InferFactoryModels<TFactory> = TFactory extends (
|
|
153
161
|
model: infer TModel,
|
|
154
|
-
...args: Array<
|
|
155
|
-
) =>
|
|
162
|
+
...args: Array<never>
|
|
163
|
+
) => unknown
|
|
156
164
|
? TModel extends string
|
|
157
165
|
? TModel
|
|
158
166
|
: string
|
|
159
167
|
: string
|
|
160
168
|
|
|
161
|
-
/**
|
|
162
|
-
* Infer the config parameter type from an adapter factory function.
|
|
163
|
-
*/
|
|
164
|
-
type InferConfig<TFactory> = TFactory extends (
|
|
165
|
-
model: any,
|
|
166
|
-
config?: infer TConfig,
|
|
167
|
-
) => any
|
|
168
|
-
? TConfig
|
|
169
|
-
: undefined
|
|
170
|
-
|
|
171
169
|
/**
|
|
172
170
|
* Infer the adapter return type from a factory function.
|
|
173
171
|
*/
|
|
174
172
|
type InferAdapterReturn<TFactory> = TFactory extends (
|
|
175
|
-
...args: Array<
|
|
173
|
+
...args: Array<never>
|
|
176
174
|
) => infer TReturn
|
|
177
175
|
? TReturn
|
|
178
176
|
: never
|
|
179
177
|
|
|
178
|
+
/**
|
|
179
|
+
* Extracts all parameter types after the model parameter from a factory,
|
|
180
|
+
* preserving labels and optionality (e.g. `[apiKey: string, config?: C]`).
|
|
181
|
+
* Note: overloaded factories resolve against their last overload (a
|
|
182
|
+
* `Parameters` limitation).
|
|
183
|
+
*/
|
|
184
|
+
type InferRestArgs<TFactory extends AnyAdapterFactory> =
|
|
185
|
+
Parameters<TFactory> extends [unknown?, ...infer TRest] ? TRest : []
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* The factory signature produced by `extendAdapter`: accepts both original
|
|
189
|
+
* and custom model names while preserving all remaining parameters and the
|
|
190
|
+
* return type of the original factory.
|
|
191
|
+
*/
|
|
192
|
+
type ExtendedFactory<
|
|
193
|
+
TFactory extends AnyAdapterFactory,
|
|
194
|
+
TDefs extends ReadonlyArray<ExtendedModelDef>,
|
|
195
|
+
> = (
|
|
196
|
+
model: InferFactoryModels<TFactory> | ExtractCustomModelNames<TDefs>,
|
|
197
|
+
...args: InferRestArgs<TFactory>
|
|
198
|
+
) => InferAdapterReturn<TFactory>
|
|
199
|
+
|
|
180
200
|
// ===========================
|
|
181
201
|
// extendAdapter Function
|
|
182
202
|
// ===========================
|
|
@@ -225,19 +245,17 @@ type InferAdapterReturn<TFactory> = TFactory extends (
|
|
|
225
245
|
* ```
|
|
226
246
|
*/
|
|
227
247
|
export function extendAdapter<
|
|
228
|
-
TFactory extends
|
|
248
|
+
TFactory extends AnyAdapterFactory,
|
|
229
249
|
const TDefs extends ReadonlyArray<ExtendedModelDef>,
|
|
230
|
-
>(
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
: [config?: InferConfig<TFactory>]
|
|
238
|
-
) => InferAdapterReturn<TFactory> {
|
|
250
|
+
>(factory: TFactory, _customModels: TDefs): ExtendedFactory<TFactory, TDefs>
|
|
251
|
+
// The implementation signature stays at the honest `AnyAdapterFactory` width;
|
|
252
|
+
// the overload above performs the deliberate model-union widening.
|
|
253
|
+
export function extendAdapter(
|
|
254
|
+
factory: AnyAdapterFactory,
|
|
255
|
+
_customModels: ReadonlyArray<ExtendedModelDef>,
|
|
256
|
+
): AnyAdapterFactory {
|
|
239
257
|
// At runtime, we simply pass through to the original factory.
|
|
240
258
|
// The _customModels parameter is only used for type inference.
|
|
241
259
|
// No runtime validation - users are trusted to pass valid model names.
|
|
242
|
-
return factory
|
|
260
|
+
return factory
|
|
243
261
|
}
|
package/src/index.ts
CHANGED
|
@@ -52,6 +52,16 @@ export {
|
|
|
52
52
|
type InferToolOutput,
|
|
53
53
|
} from './activities/chat/tools/tool-definition'
|
|
54
54
|
|
|
55
|
+
// MCP chat option types
|
|
56
|
+
export type {
|
|
57
|
+
MCPToolSource,
|
|
58
|
+
ChatMCPOptions,
|
|
59
|
+
MCPConnectionPolicy,
|
|
60
|
+
} from './activities/chat/mcp/types'
|
|
61
|
+
|
|
62
|
+
// MCP error classes (value exports — usable with instanceof)
|
|
63
|
+
export { MCPDuplicateToolNameError } from './activities/chat/mcp/manager'
|
|
64
|
+
|
|
55
65
|
// Schema conversion (Standard JSON Schema compliant)
|
|
56
66
|
export {
|
|
57
67
|
convertSchemaToJsonSchema,
|
|
@@ -104,4 +104,22 @@ export class InternalLogger {
|
|
|
104
104
|
request(message: string, meta?: Record<string, unknown>): void {
|
|
105
105
|
this.emit('debug', 'request', message, meta)
|
|
106
106
|
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Log a non-fatal misconfiguration or recoverable anomaly. Gated by the
|
|
110
|
+
* `errors` category — on by default (and when `debug` is unspecified), so
|
|
111
|
+
* silent-drop conditions surface, but still silenced by `debug: false`,
|
|
112
|
+
* which honors the "disable everything including errors" contract. Routes to
|
|
113
|
+
* the underlying logger's `warn` level.
|
|
114
|
+
*/
|
|
115
|
+
warn(message: string, meta?: Record<string, unknown>): void {
|
|
116
|
+
if (!this.categories.errors) return
|
|
117
|
+
const prefixed = `⚠️ [tanstack-ai:warn] ⚠️ ${message}`
|
|
118
|
+
try {
|
|
119
|
+
this.logger.warn(prefixed, meta)
|
|
120
|
+
} catch {
|
|
121
|
+
// User-supplied logger threw; swallow so a broken logger never masks the
|
|
122
|
+
// condition we were trying to surface.
|
|
123
|
+
}
|
|
124
|
+
}
|
|
107
125
|
}
|
package/src/middlewares/otel.ts
CHANGED
|
@@ -4,6 +4,10 @@ import {
|
|
|
4
4
|
context as otelContext,
|
|
5
5
|
trace as otelTrace,
|
|
6
6
|
} from '@opentelemetry/api'
|
|
7
|
+
import {
|
|
8
|
+
MAX_TOKENS_KEYS,
|
|
9
|
+
NESTED_MAX_TOKENS_KEY,
|
|
10
|
+
} from '../utilities/sampling-keys'
|
|
7
11
|
import type {
|
|
8
12
|
AttributeValue,
|
|
9
13
|
Exception,
|
|
@@ -162,6 +166,19 @@ function messageEventName(role: string): string {
|
|
|
162
166
|
}
|
|
163
167
|
}
|
|
164
168
|
|
|
169
|
+
/**
|
|
170
|
+
* Return the first candidate that is a finite `number`, or `undefined`. Used to
|
|
171
|
+
* pick a sampling attribute from among the several provider-native spellings.
|
|
172
|
+
*/
|
|
173
|
+
function firstNumber(...candidates: Array<unknown>): number | undefined {
|
|
174
|
+
for (const candidate of candidates) {
|
|
175
|
+
if (typeof candidate === 'number' && Number.isFinite(candidate)) {
|
|
176
|
+
return candidate
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
return undefined
|
|
180
|
+
}
|
|
181
|
+
|
|
165
182
|
function errorMessage(err: unknown): string | undefined {
|
|
166
183
|
if (err instanceof Error) return err.message
|
|
167
184
|
if (typeof err === 'string') return err
|
|
@@ -333,12 +350,37 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
|
|
|
333
350
|
'gen_ai.request.model': ctx.model,
|
|
334
351
|
'tanstack.ai.iteration': ctx.iteration,
|
|
335
352
|
}
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
353
|
+
// Sampling options now live in provider-native `modelOptions`, and
|
|
354
|
+
// providers spell them differently (e.g. `max_output_tokens`,
|
|
355
|
+
// `max_completion_tokens`, `maxOutputTokens`, `num_predict`). Read the
|
|
356
|
+
// first numeric value among the known spellings — including Ollama's
|
|
357
|
+
// nested `options` — so gen_ai attributes populate across providers.
|
|
358
|
+
const sampling = config.modelOptions ?? {}
|
|
359
|
+
const nestedOptions =
|
|
360
|
+
sampling['options'] && typeof sampling['options'] === 'object'
|
|
361
|
+
? (sampling['options'] as Record<string, unknown>)
|
|
362
|
+
: undefined
|
|
363
|
+
const samplingTemperature = firstNumber(
|
|
364
|
+
sampling['temperature'],
|
|
365
|
+
nestedOptions?.['temperature'],
|
|
366
|
+
)
|
|
367
|
+
const samplingTopP = firstNumber(
|
|
368
|
+
sampling['top_p'],
|
|
369
|
+
sampling['topP'],
|
|
370
|
+
nestedOptions?.['top_p'],
|
|
371
|
+
)
|
|
372
|
+
// Spellings come from the shared `MAX_TOKENS_KEYS` table so this stays
|
|
373
|
+
// in lockstep with the summarize wrapper's caller-limit detection.
|
|
374
|
+
const samplingMaxTokens = firstNumber(
|
|
375
|
+
...MAX_TOKENS_KEYS.map((k) => sampling[k]),
|
|
376
|
+
nestedOptions?.[NESTED_MAX_TOKENS_KEY],
|
|
377
|
+
)
|
|
378
|
+
if (samplingTemperature !== undefined)
|
|
379
|
+
baseAttrs['gen_ai.request.temperature'] = samplingTemperature
|
|
380
|
+
if (samplingTopP !== undefined)
|
|
381
|
+
baseAttrs['gen_ai.request.top_p'] = samplingTopP
|
|
382
|
+
if (samplingMaxTokens !== undefined)
|
|
383
|
+
baseAttrs['gen_ai.request.max_tokens'] = samplingMaxTokens
|
|
342
384
|
|
|
343
385
|
const baseOptions: SpanOptions = {
|
|
344
386
|
kind: SpanKind.CLIENT,
|
package/src/types.ts
CHANGED
|
@@ -490,6 +490,12 @@ export type ToolExecutionContext<TContext = unknown> =
|
|
|
490
490
|
RuntimeContextField<TContext> & {
|
|
491
491
|
/** The ID of the tool call being executed */
|
|
492
492
|
toolCallId?: string
|
|
493
|
+
/**
|
|
494
|
+
* Abort signal for the current chat run. Aborts when the run's
|
|
495
|
+
* `abortController` fires (or middleware aborts). Long-running tools —
|
|
496
|
+
* e.g. MCP `callTool` — should forward this to cancel in-flight work.
|
|
497
|
+
*/
|
|
498
|
+
abortSignal?: AbortSignal
|
|
493
499
|
/**
|
|
494
500
|
* Emit a custom event during tool execution.
|
|
495
501
|
* Events are streamed to the client in real-time as AG-UI CUSTOM events.
|
|
@@ -812,41 +818,6 @@ export interface TextOptions<
|
|
|
812
818
|
*/
|
|
813
819
|
systemPrompts?: Array<SystemPrompt>
|
|
814
820
|
agentLoopStrategy?: AgentLoopStrategy
|
|
815
|
-
/**
|
|
816
|
-
* Controls the randomness of the output.
|
|
817
|
-
* Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
|
|
818
|
-
* Range: [0.0, 2.0]
|
|
819
|
-
*
|
|
820
|
-
* Note: Generally recommended to use either temperature or topP, but not both.
|
|
821
|
-
*
|
|
822
|
-
* Provider usage:
|
|
823
|
-
* - OpenAI: `temperature` (number) - in text.top_p field
|
|
824
|
-
* - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
|
|
825
|
-
* - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
|
|
826
|
-
*/
|
|
827
|
-
temperature?: number
|
|
828
|
-
/**
|
|
829
|
-
* Nucleus sampling parameter. An alternative to temperature sampling.
|
|
830
|
-
* The model considers the results of tokens with topP probability mass.
|
|
831
|
-
* For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
|
|
832
|
-
*
|
|
833
|
-
* Note: Generally recommended to use either temperature or topP, but not both.
|
|
834
|
-
*
|
|
835
|
-
* Provider usage:
|
|
836
|
-
* - OpenAI: `text.top_p` (number)
|
|
837
|
-
* - Anthropic: `top_p` (number | null)
|
|
838
|
-
* - Gemini: `generationConfig.topP` (number)
|
|
839
|
-
*/
|
|
840
|
-
topP?: number
|
|
841
|
-
/**
|
|
842
|
-
* The maximum number of tokens to generate in the response.
|
|
843
|
-
*
|
|
844
|
-
* Provider usage:
|
|
845
|
-
* - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
|
|
846
|
-
* - Anthropic: `max_tokens` (number, required) - range x >= 1
|
|
847
|
-
* - Gemini: `generationConfig.maxOutputTokens` (number)
|
|
848
|
-
*/
|
|
849
|
-
maxTokens?: number
|
|
850
821
|
/**
|
|
851
822
|
* Additional metadata to attach to the request.
|
|
852
823
|
* Can be used for tracking, debugging, or passing custom information.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Single source of truth for the provider-native key spellings that cap output
|
|
3
|
+
* tokens. Sampling options live in opaque, provider-native `modelOptions`, and
|
|
4
|
+
* every provider spells the token cap differently. Two call sites must agree on
|
|
5
|
+
* this set or they silently drift:
|
|
6
|
+
*
|
|
7
|
+
* - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied
|
|
8
|
+
* token limit so the summarize default never overrides it.
|
|
9
|
+
* - `middlewares/otel.ts` — picks the first numeric spelling to populate the
|
|
10
|
+
* `gen_ai.request.max_tokens` attribute across providers.
|
|
11
|
+
*
|
|
12
|
+
* Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →
|
|
13
|
+
* native-key map) in the summarize wrapper.
|
|
14
|
+
*/
|
|
15
|
+
export const MAX_TOKENS_KEYS = [
|
|
16
|
+
'max_output_tokens', // OpenAI (Responses)
|
|
17
|
+
'max_tokens', // Anthropic / Grok
|
|
18
|
+
'max_completion_tokens', // Groq
|
|
19
|
+
'maxOutputTokens', // Gemini
|
|
20
|
+
'maxCompletionTokens', // OpenRouter
|
|
21
|
+
'maxTokens', // generic / migration leftover (no adapter reads it)
|
|
22
|
+
] as const
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Ollama nests sampling under `options`; its token cap is `options.num_predict`
|
|
26
|
+
* rather than a flat key.
|
|
27
|
+
*/
|
|
28
|
+
export const NESTED_MAX_TOKENS_KEY = 'num_predict' as const
|