@tanstack/ai 0.26.1 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/dist/esm/activities/chat/index.d.ts +7 -6
  2. package/dist/esm/activities/chat/index.js +78 -27
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/mcp/manager.d.ts +25 -0
  5. package/dist/esm/activities/chat/mcp/manager.js +71 -0
  6. package/dist/esm/activities/chat/mcp/manager.js.map +1 -0
  7. package/dist/esm/activities/chat/mcp/types.d.ts +56 -0
  8. package/dist/esm/activities/chat/middleware/types.d.ts +1 -4
  9. package/dist/esm/activities/chat/stream/message-updaters.js +20 -8
  10. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -1
  11. package/dist/esm/activities/chat/tools/tool-calls.d.ts +1 -1
  12. package/dist/esm/activities/chat/tools/tool-calls.js +2 -1
  13. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  14. package/dist/esm/activities/summarize/chat-stream-summarize.js +62 -3
  15. package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
  16. package/dist/esm/extend-adapter.d.ts +22 -6
  17. package/dist/esm/extend-adapter.js.map +1 -1
  18. package/dist/esm/index.d.ts +2 -0
  19. package/dist/esm/index.js +2 -0
  20. package/dist/esm/index.js.map +1 -1
  21. package/dist/esm/logger/internal-logger.d.ts +8 -0
  22. package/dist/esm/logger/internal-logger.js +15 -0
  23. package/dist/esm/logger/internal-logger.js.map +1 -1
  24. package/dist/esm/middlewares/otel.js +30 -6
  25. package/dist/esm/middlewares/otel.js.map +1 -1
  26. package/dist/esm/types.d.ts +6 -35
  27. package/dist/esm/utilities/sampling-keys.d.ts +20 -0
  28. package/dist/esm/utilities/sampling-keys.js +20 -0
  29. package/dist/esm/utilities/sampling-keys.js.map +1 -0
  30. package/package.json +2 -2
  31. package/skills/ai-core/adapter-configuration/SKILL.md +67 -6
  32. package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +6 -3
  33. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +3 -0
  34. package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +10 -1
  35. package/skills/ai-core/adapter-configuration/references/openai-adapter.md +4 -0
  36. package/skills/ai-core/chat-experience/SKILL.md +95 -7
  37. package/skills/ai-core/middleware/SKILL.md +11 -0
  38. package/skills/ai-core/tool-calling/SKILL.md +287 -0
  39. package/src/activities/chat/index.ts +97 -35
  40. package/src/activities/chat/mcp/manager.ts +85 -0
  41. package/src/activities/chat/mcp/types.ts +66 -0
  42. package/src/activities/chat/middleware/types.ts +1 -4
  43. package/src/activities/chat/stream/message-updaters.ts +22 -9
  44. package/src/activities/chat/tools/tool-calls.ts +2 -0
  45. package/src/activities/summarize/chat-stream-summarize.ts +162 -3
  46. package/src/extend-adapter.ts +42 -24
  47. package/src/index.ts +10 -0
  48. package/src/logger/internal-logger.ts +18 -0
  49. package/src/middlewares/otel.ts +48 -6
  50. package/src/types.ts +6 -35
  51. package/src/utilities/sampling-keys.ts +28 -0
@@ -0,0 +1,66 @@
1
+ import type { ServerTool } from '../tools/tool-definition'
2
+
3
+ /**
4
+ * Minimal structural shape that `chat({ mcp })` needs from an MCP client.
5
+ *
6
+ * `@tanstack/ai-mcp`'s `MCPClient` and `MCPClients` satisfy this interface by
7
+ * shape — the core `@tanstack/ai` package does NOT import `@tanstack/ai-mcp`
8
+ * (ai-mcp depends on ai, not the reverse).
9
+ */
10
+ export interface MCPToolSource {
11
+ // Keep the options shape in sync with ai-mcp's `ToolsOptions` — extra
12
+ // optional fields added there still match structurally, but chat() only
13
+ // forwards what is declared here.
14
+ tools: (options?: { lazy?: boolean }) => Promise<Array<ServerTool>>
15
+ close: () => Promise<void>
16
+ }
17
+
18
+ /**
19
+ * Controls what happens to MCP connections when the chat run ends.
20
+ *
21
+ * - `'close'` (default) — `chat()` closes each connection when the run ends
22
+ * (after the agent loop completes and the stream is drained), so tools can
23
+ * still execute throughout the run.
24
+ * - `'keep-alive'` — `chat()` never closes the connections; the caller owns
25
+ * their lifecycle (e.g. keep them warm across requests).
26
+ */
27
+ export type MCPConnectionPolicy = 'close' | 'keep-alive'
28
+
29
+ /**
30
+ * Options controlling MCP tool discovery and lifecycle for a `chat()` call.
31
+ */
32
+ export interface ChatMCPOptions {
33
+ /**
34
+ * The MCP clients or client pools to discover tools from and manage.
35
+ */
36
+ clients: Array<MCPToolSource>
37
+
38
+ /**
39
+ * Connection lifecycle policy applied to all clients when the run ends.
40
+ *
41
+ * Defaults to `'close'`.
42
+ */
43
+ connection?: MCPConnectionPolicy
44
+
45
+ /**
46
+ * When `true`, tool schemas are fetched lazily (forwarded to
47
+ * `tools({ lazy: true })`).
48
+ *
49
+ * Defaults to `false`.
50
+ */
51
+ lazyTools?: boolean
52
+
53
+ /**
54
+ * Called when tool discovery fails for a single source.
55
+ *
56
+ * - Throw (or re-throw) from this handler to fail the entire chat call fast.
57
+ * - Return normally to skip that source and continue with remaining clients.
58
+ * - Omit this handler entirely to rethrow the error (fail-fast by default).
59
+ *
60
+ * Async handlers are awaited, so a rejected promise also fails fast.
61
+ */
62
+ onDiscoveryError?: (
63
+ error: unknown,
64
+ source: MCPToolSource,
65
+ ) => void | Promise<void>
66
+ }
@@ -90,7 +90,7 @@ export interface ChatMiddlewareContext<TContext = unknown> {
90
90
  systemPrompts: Array<SystemPrompt>
91
91
  /** Names of configured tools, if any */
92
92
  toolNames?: Array<string>
93
- /** Flattened generation options (temperature, topP, maxTokens, metadata) */
93
+ /** Flattened generation options (metadata) */
94
94
  options?: Record<string, unknown> | undefined
95
95
  /** Provider-specific model options */
96
96
  modelOptions?: Record<string, unknown> | undefined
@@ -130,9 +130,6 @@ export interface ChatMiddlewareConfig {
130
130
  messages: Array<ModelMessage>
131
131
  systemPrompts: Array<SystemPrompt>
132
132
  tools: Array<Tool>
133
- temperature?: number
134
- topP?: number
135
- maxTokens?: number
136
133
  metadata?: Record<string, unknown> | undefined
137
134
  modelOptions?: Record<string, unknown> | undefined
138
135
  }
@@ -161,10 +161,14 @@ export function updateToolCallApproval(
161
161
  )
162
162
 
163
163
  if (toolCallPart) {
164
- toolCallPart.state = 'approval-requested'
165
- toolCallPart.approval = {
166
- id: approvalId,
167
- needsApproval: true,
164
+ const index = parts.indexOf(toolCallPart)
165
+ parts[index] = {
166
+ ...toolCallPart,
167
+ state: 'approval-requested',
168
+ approval: {
169
+ id: approvalId,
170
+ needsApproval: true,
171
+ },
168
172
  }
169
173
  }
170
174
 
@@ -192,7 +196,8 @@ export function updateToolCallState(
192
196
  )
193
197
 
194
198
  if (toolCallPart) {
195
- toolCallPart.state = state
199
+ const index = parts.indexOf(toolCallPart)
200
+ parts[index] = { ...toolCallPart, state }
196
201
  }
197
202
 
198
203
  return { ...msg, parts }
@@ -217,8 +222,12 @@ export function updateToolCallWithOutput(
217
222
  )
218
223
 
219
224
  if (toolCallPart) {
220
- toolCallPart.output = errorText ? { error: errorText } : output
221
- toolCallPart.state = state ?? (errorText ? 'input-complete' : 'complete')
225
+ const index = parts.indexOf(toolCallPart)
226
+ parts[index] = {
227
+ ...toolCallPart,
228
+ output: errorText ? { error: errorText } : output,
229
+ state: state ?? (errorText ? 'input-complete' : 'complete'),
230
+ }
222
231
  }
223
232
 
224
233
  return { ...msg, parts }
@@ -242,8 +251,12 @@ export function updateToolCallApprovalResponse(
242
251
  )
243
252
 
244
253
  if (toolCallPart && toolCallPart.approval) {
245
- toolCallPart.approval.approved = approved
246
- toolCallPart.state = 'approval-responded'
254
+ const index = parts.indexOf(toolCallPart)
255
+ parts[index] = {
256
+ ...toolCallPart,
257
+ approval: { ...toolCallPart.approval, approved },
258
+ state: 'approval-responded',
259
+ }
247
260
  }
248
261
 
249
262
  return { ...msg, parts }
@@ -599,6 +599,7 @@ export async function* executeToolCalls<TContext = unknown>(
599
599
  ) => CustomEvent,
600
600
  middlewareHooks?: ToolExecutionMiddlewareHooks,
601
601
  userContext?: TContext,
602
+ abortSignal?: AbortSignal,
602
603
  ): AsyncGenerator<CustomEvent, ExecuteToolCallsResult, void> {
603
604
  const results: Array<ToolResult> = []
604
605
  const needsApproval: Array<ApprovalRequest> = []
@@ -679,6 +680,7 @@ export async function* executeToolCalls<TContext = unknown>(
679
680
  const context = {
680
681
  toolCallId: toolCall.id,
681
682
  context: userContext,
683
+ abortSignal,
682
684
  emitCustomEvent: (eventName: string, value: Record<string, any>) => {
683
685
  if (createCustomEventChunk) {
684
686
  pendingEvents.push(
@@ -1,5 +1,6 @@
1
1
  import { EventType } from '@ag-ui/core'
2
2
  import { toRunErrorPayload } from '../error-payload'
3
+ import { MAX_TOKENS_KEYS } from '../../utilities/sampling-keys'
3
4
  import { BaseSummarizeAdapter } from './adapter'
4
5
  import type {
5
6
  StreamChunk,
@@ -23,6 +24,139 @@ export interface ChatStreamCapable {
23
24
  chatStream: (options: TextOptions<any>) => AsyncIterable<StreamChunk>
24
25
  }
25
26
 
27
+ /**
28
+ * Provider-native max-output-tokens key per summarize-adapter `name`. summarize
29
+ * is provider-agnostic and forwards `modelOptions` opaquely to the wrapped text
30
+ * adapter, so `maxLength` must be written under the exact key the underlying
31
+ * provider reads — no adapter reads a generic `maxTokens`. Ollama is the one
32
+ * exception: it nests sampling under `options`, so it has no entry here and is
33
+ * handled as a special nested case in `applyMaxLength`/`applyDefaultTemperature`.
34
+ *
35
+ * Keep in sync with each adapter's wire mapping:
36
+ * - OpenAI (Responses): `max_output_tokens`
37
+ * - Anthropic / Grok: `max_tokens`
38
+ * - Groq: `max_completion_tokens`
39
+ * - Gemini: `maxOutputTokens`
40
+ * - OpenRouter: `maxCompletionTokens`
41
+ * - Ollama: nested `options.num_predict` (no entry — see `applyMaxLength`)
42
+ */
43
+ const MAX_TOKENS_KEY_BY_ADAPTER: Record<string, string> = {
44
+ openai: 'max_output_tokens',
45
+ anthropic: 'max_tokens',
46
+ grok: 'max_tokens',
47
+ groq: 'max_completion_tokens',
48
+ gemini: 'maxOutputTokens',
49
+ openrouter: 'maxCompletionTokens',
50
+ }
51
+
52
+ /**
53
+ * Every flat key any supported provider uses to cap output tokens (plus the
54
+ * generic `maxTokens` spelling no adapter reads). Used to detect a
55
+ * caller-supplied token limit so the summarize default never overrides an
56
+ * explicit caller value. Shared with the OTel middleware via
57
+ * `MAX_TOKENS_KEYS` so the two spelling sets cannot drift.
58
+ */
59
+ const KNOWN_MAX_TOKENS_KEYS = MAX_TOKENS_KEYS
60
+
61
+ /**
62
+ * Whether `applyMaxLength` knows how to place a token limit for this adapter
63
+ * `name` (either the nested Ollama shape or a flat provider-native key).
64
+ * Used to surface a warning when `maxLength` would otherwise be silently
65
+ * dropped for an unrecognised adapter name.
66
+ */
67
+ function isKnownMaxTokensAdapter(adapterName: string): boolean {
68
+ return (
69
+ adapterName === 'ollama' ||
70
+ MAX_TOKENS_KEY_BY_ADAPTER[adapterName] !== undefined
71
+ )
72
+ }
73
+
74
+ /**
75
+ * Apply the low-temperature summarize default to a working copy of the
76
+ * caller's `modelOptions`, placed where the wrapped provider actually reads
77
+ * it (nested under `options` for Ollama, flat otherwise). The caller always
78
+ * wins: if they already set `temperature` in that location, it is untouched.
79
+ */
80
+ function applyDefaultTemperature(
81
+ adapterName: string,
82
+ temperature: number,
83
+ modelOptions: Record<string, unknown>,
84
+ ): Record<string, unknown> {
85
+ const merged: Record<string, unknown> = { ...modelOptions }
86
+
87
+ if (adapterName === 'ollama') {
88
+ const existing =
89
+ merged.options && typeof merged.options === 'object'
90
+ ? (merged.options as Record<string, unknown>)
91
+ : undefined
92
+ if (existing && 'temperature' in existing) return merged
93
+ merged.options = { temperature, ...existing }
94
+ return merged
95
+ }
96
+
97
+ if ('temperature' in merged) return merged
98
+ merged.temperature = temperature
99
+ return merged
100
+ }
101
+
102
+ /**
103
+ * Resolve `maxLength` to the provider-native max-output-tokens key for the
104
+ * given summarize-adapter `name` (this wrapper's OWN `name`, not the wrapped
105
+ * text adapter's) and merge it into a working copy of the caller's
106
+ * `modelOptions`. The caller always wins: if they already set any recognised
107
+ * token-limit key (flat or, for Ollama, nested `options.num_predict`), the
108
+ * default is left untouched. Unknown/unrecognised adapter names fall back to
109
+ * NOT setting a token key (the prompt hint still asks the model to stay under
110
+ * `maxLength`) rather than writing a dead key no provider reads.
111
+ *
112
+ * Caveat (intentional): "caller wins" keys off ANY recognised spelling in
113
+ * `KNOWN_MAX_TOKENS_KEYS`, but only the adapter's native key is read on the
114
+ * wire. So a caller who sets a NON-native spelling for this provider — e.g.
115
+ * `maxTokens`, or Anthropic's `max_tokens` against an OpenAI adapter — suppresses
116
+ * the summarize default WITHOUT getting their own value applied either: neither
117
+ * cap reaches the wire. This favours never clobbering a migration leftover over
118
+ * guaranteeing a cap; the prompt-level hint still asks the model to stay under
119
+ * `maxLength`. Rename the key to the provider-native spelling to forward it.
120
+ */
121
+ function applyMaxLength(
122
+ adapterName: string,
123
+ maxLength: number,
124
+ modelOptions: Record<string, unknown>,
125
+ ): Record<string, unknown> {
126
+ const merged: Record<string, unknown> = { ...modelOptions }
127
+
128
+ if (adapterName === 'ollama') {
129
+ // Honor a caller-set limit in either shape: a recognised flat key (e.g.
130
+ // left over from a migration) or the nested `options.num_predict`.
131
+ const callerSetFlatLimit = KNOWN_MAX_TOKENS_KEYS.some(
132
+ (k) => typeof merged[k] === 'number',
133
+ )
134
+ const existing =
135
+ merged.options && typeof merged.options === 'object'
136
+ ? (merged.options as Record<string, unknown>)
137
+ : undefined
138
+ if (
139
+ callerSetFlatLimit ||
140
+ (existing && typeof existing.num_predict === 'number')
141
+ ) {
142
+ return merged
143
+ }
144
+ merged.options = { num_predict: maxLength, ...existing }
145
+ return merged
146
+ }
147
+
148
+ const key = MAX_TOKENS_KEY_BY_ADAPTER[adapterName]
149
+ if (key === undefined) return merged
150
+
151
+ const callerSetLimit = KNOWN_MAX_TOKENS_KEYS.some(
152
+ (k) => typeof merged[k] === 'number',
153
+ )
154
+ if (callerSetLimit) return merged
155
+
156
+ merged[key] = maxLength
157
+ return merged
158
+ }
159
+
26
160
  /**
27
161
  * Extract the per-model `modelOptions` type a text adapter accepts. Used by
28
162
  * provider summarize factories so their `modelOptions` IntelliSense matches
@@ -195,13 +329,38 @@ export class ChatStreamSummarizeAdapter<
195
329
  options: SummarizationOptions<TProviderOptions>,
196
330
  systemPrompt: string,
197
331
  ): TextOptions<TProviderOptions> {
332
+ // Sampling knobs now live in provider-native `modelOptions`. Apply the
333
+ // low-temperature default where the wrapped provider actually reads it
334
+ // (nested under `options` for Ollama, flat otherwise) so callers can still
335
+ // override it. Resolving the placement from this summarize adapter's OWN
336
+ // `name` keeps the default off the wire correctly per provider — a flat
337
+ // `temperature` would be silently dropped by Ollama while still showing up
338
+ // in OTel.
339
+ let working: Record<string, unknown> = {
340
+ ...(options.modelOptions as Record<string, unknown> | undefined),
341
+ }
342
+ working = applyDefaultTemperature(this.name, 0.3, working)
343
+ // `maxLength` must reach the wire under the provider-native token key (it
344
+ // differs per provider, and no adapter reads a generic `maxTokens`).
345
+ // Resolve it from this summarize adapter's `name` (the constructor arg,
346
+ // not the wrapped text adapter's name), never overriding a caller-supplied
347
+ // token limit.
348
+ if (options.maxLength !== undefined) {
349
+ if (!isKnownMaxTokensAdapter(this.name)) {
350
+ options.logger.warn(
351
+ `summarize: maxLength=${options.maxLength} could not be mapped to a provider token key for adapter name "${this.name}" — it was dropped from modelOptions (the prompt still asks the model to stay under it). Construct ChatStreamSummarizeAdapter with a recognised provider name to forward the cap.`,
352
+ { provider: this.name },
353
+ )
354
+ }
355
+ working = applyMaxLength(this.name, options.maxLength, working)
356
+ }
357
+ const modelOptions = working as TProviderOptions
358
+
198
359
  return {
199
360
  model: options.model,
200
361
  messages: [{ role: 'user', content: options.text }],
201
362
  systemPrompts: [systemPrompt],
202
- maxTokens: options.maxLength,
203
- temperature: 0.3,
204
- modelOptions: options.modelOptions,
363
+ modelOptions,
205
364
  logger: options.logger,
206
365
  }
207
366
  }
@@ -144,6 +144,14 @@ type ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =
144
144
  // Factory Type Inference
145
145
  // ===========================
146
146
 
147
+ /**
148
+ * The widest factory shape `extendAdapter` accepts: any function taking a
149
+ * model as its first parameter. Parameters are contravariant, so `never`
150
+ * params and an `unknown` return accept every factory without resorting
151
+ * to `any`.
152
+ */
153
+ type AnyAdapterFactory = (model: never, ...args: Array<never>) => unknown
154
+
147
155
  /**
148
156
  * Infer the model parameter type from an adapter factory function.
149
157
  * For generic functions like `<T extends Union>(model: T)`, this gets `T` which
@@ -151,32 +159,44 @@ type ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =
151
159
  */
152
160
  type InferFactoryModels<TFactory> = TFactory extends (
153
161
  model: infer TModel,
154
- ...args: Array<any>
155
- ) => any
162
+ ...args: Array<never>
163
+ ) => unknown
156
164
  ? TModel extends string
157
165
  ? TModel
158
166
  : string
159
167
  : string
160
168
 
161
- /**
162
- * Infer the config parameter type from an adapter factory function.
163
- */
164
- type InferConfig<TFactory> = TFactory extends (
165
- model: any,
166
- config?: infer TConfig,
167
- ) => any
168
- ? TConfig
169
- : undefined
170
-
171
169
  /**
172
170
  * Infer the adapter return type from a factory function.
173
171
  */
174
172
  type InferAdapterReturn<TFactory> = TFactory extends (
175
- ...args: Array<any>
173
+ ...args: Array<never>
176
174
  ) => infer TReturn
177
175
  ? TReturn
178
176
  : never
179
177
 
178
+ /**
179
+ * Extracts all parameter types after the model parameter from a factory,
180
+ * preserving labels and optionality (e.g. `[apiKey: string, config?: C]`).
181
+ * Note: overloaded factories resolve against their last overload (a
182
+ * `Parameters` limitation).
183
+ */
184
+ type InferRestArgs<TFactory extends AnyAdapterFactory> =
185
+ Parameters<TFactory> extends [unknown?, ...infer TRest] ? TRest : []
186
+
187
+ /**
188
+ * The factory signature produced by `extendAdapter`: accepts both original
189
+ * and custom model names while preserving all remaining parameters and the
190
+ * return type of the original factory.
191
+ */
192
+ type ExtendedFactory<
193
+ TFactory extends AnyAdapterFactory,
194
+ TDefs extends ReadonlyArray<ExtendedModelDef>,
195
+ > = (
196
+ model: InferFactoryModels<TFactory> | ExtractCustomModelNames<TDefs>,
197
+ ...args: InferRestArgs<TFactory>
198
+ ) => InferAdapterReturn<TFactory>
199
+
180
200
  // ===========================
181
201
  // extendAdapter Function
182
202
  // ===========================
@@ -225,19 +245,17 @@ type InferAdapterReturn<TFactory> = TFactory extends (
225
245
  * ```
226
246
  */
227
247
  export function extendAdapter<
228
- TFactory extends (...args: Array<any>) => any,
248
+ TFactory extends AnyAdapterFactory,
229
249
  const TDefs extends ReadonlyArray<ExtendedModelDef>,
230
- >(
231
- factory: TFactory,
232
- _customModels: TDefs,
233
- ): (
234
- model: InferFactoryModels<TFactory> | ExtractCustomModelNames<TDefs>,
235
- ...args: InferConfig<TFactory> extends undefined
236
- ? []
237
- : [config?: InferConfig<TFactory>]
238
- ) => InferAdapterReturn<TFactory> {
250
+ >(factory: TFactory, _customModels: TDefs): ExtendedFactory<TFactory, TDefs>
251
+ // The implementation signature stays at the honest `AnyAdapterFactory` width;
252
+ // the overload above performs the deliberate model-union widening.
253
+ export function extendAdapter(
254
+ factory: AnyAdapterFactory,
255
+ _customModels: ReadonlyArray<ExtendedModelDef>,
256
+ ): AnyAdapterFactory {
239
257
  // At runtime, we simply pass through to the original factory.
240
258
  // The _customModels parameter is only used for type inference.
241
259
  // No runtime validation - users are trusted to pass valid model names.
242
- return factory as any
260
+ return factory
243
261
  }
package/src/index.ts CHANGED
@@ -52,6 +52,16 @@ export {
52
52
  type InferToolOutput,
53
53
  } from './activities/chat/tools/tool-definition'
54
54
 
55
+ // MCP chat option types
56
+ export type {
57
+ MCPToolSource,
58
+ ChatMCPOptions,
59
+ MCPConnectionPolicy,
60
+ } from './activities/chat/mcp/types'
61
+
62
+ // MCP error classes (value exports — usable with instanceof)
63
+ export { MCPDuplicateToolNameError } from './activities/chat/mcp/manager'
64
+
55
65
  // Schema conversion (Standard JSON Schema compliant)
56
66
  export {
57
67
  convertSchemaToJsonSchema,
@@ -104,4 +104,22 @@ export class InternalLogger {
104
104
  request(message: string, meta?: Record<string, unknown>): void {
105
105
  this.emit('debug', 'request', message, meta)
106
106
  }
107
+
108
+ /**
109
+ * Log a non-fatal misconfiguration or recoverable anomaly. Gated by the
110
+ * `errors` category — on by default (and when `debug` is unspecified), so
111
+ * silent-drop conditions surface, but still silenced by `debug: false`,
112
+ * which honors the "disable everything including errors" contract. Routes to
113
+ * the underlying logger's `warn` level.
114
+ */
115
+ warn(message: string, meta?: Record<string, unknown>): void {
116
+ if (!this.categories.errors) return
117
+ const prefixed = `⚠️ [tanstack-ai:warn] ⚠️ ${message}`
118
+ try {
119
+ this.logger.warn(prefixed, meta)
120
+ } catch {
121
+ // User-supplied logger threw; swallow so a broken logger never masks the
122
+ // condition we were trying to surface.
123
+ }
124
+ }
107
125
  }
@@ -4,6 +4,10 @@ import {
4
4
  context as otelContext,
5
5
  trace as otelTrace,
6
6
  } from '@opentelemetry/api'
7
+ import {
8
+ MAX_TOKENS_KEYS,
9
+ NESTED_MAX_TOKENS_KEY,
10
+ } from '../utilities/sampling-keys'
7
11
  import type {
8
12
  AttributeValue,
9
13
  Exception,
@@ -162,6 +166,19 @@ function messageEventName(role: string): string {
162
166
  }
163
167
  }
164
168
 
169
+ /**
170
+ * Return the first candidate that is a finite `number`, or `undefined`. Used to
171
+ * pick a sampling attribute from among the several provider-native spellings.
172
+ */
173
+ function firstNumber(...candidates: Array<unknown>): number | undefined {
174
+ for (const candidate of candidates) {
175
+ if (typeof candidate === 'number' && Number.isFinite(candidate)) {
176
+ return candidate
177
+ }
178
+ }
179
+ return undefined
180
+ }
181
+
165
182
  function errorMessage(err: unknown): string | undefined {
166
183
  if (err instanceof Error) return err.message
167
184
  if (typeof err === 'string') return err
@@ -333,12 +350,37 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
333
350
  'gen_ai.request.model': ctx.model,
334
351
  'tanstack.ai.iteration': ctx.iteration,
335
352
  }
336
- if (config.temperature !== undefined)
337
- baseAttrs['gen_ai.request.temperature'] = config.temperature
338
- if (config.topP !== undefined)
339
- baseAttrs['gen_ai.request.top_p'] = config.topP
340
- if (config.maxTokens !== undefined)
341
- baseAttrs['gen_ai.request.max_tokens'] = config.maxTokens
353
+ // Sampling options now live in provider-native `modelOptions`, and
354
+ // providers spell them differently (e.g. `max_output_tokens`,
355
+ // `max_completion_tokens`, `maxOutputTokens`, `num_predict`). Read the
356
+ // first numeric value among the known spellings — including Ollama's
357
+ // nested `options` — so gen_ai attributes populate across providers.
358
+ const sampling = config.modelOptions ?? {}
359
+ const nestedOptions =
360
+ sampling['options'] && typeof sampling['options'] === 'object'
361
+ ? (sampling['options'] as Record<string, unknown>)
362
+ : undefined
363
+ const samplingTemperature = firstNumber(
364
+ sampling['temperature'],
365
+ nestedOptions?.['temperature'],
366
+ )
367
+ const samplingTopP = firstNumber(
368
+ sampling['top_p'],
369
+ sampling['topP'],
370
+ nestedOptions?.['top_p'],
371
+ )
372
+ // Spellings come from the shared `MAX_TOKENS_KEYS` table so this stays
373
+ // in lockstep with the summarize wrapper's caller-limit detection.
374
+ const samplingMaxTokens = firstNumber(
375
+ ...MAX_TOKENS_KEYS.map((k) => sampling[k]),
376
+ nestedOptions?.[NESTED_MAX_TOKENS_KEY],
377
+ )
378
+ if (samplingTemperature !== undefined)
379
+ baseAttrs['gen_ai.request.temperature'] = samplingTemperature
380
+ if (samplingTopP !== undefined)
381
+ baseAttrs['gen_ai.request.top_p'] = samplingTopP
382
+ if (samplingMaxTokens !== undefined)
383
+ baseAttrs['gen_ai.request.max_tokens'] = samplingMaxTokens
342
384
 
343
385
  const baseOptions: SpanOptions = {
344
386
  kind: SpanKind.CLIENT,
package/src/types.ts CHANGED
@@ -490,6 +490,12 @@ export type ToolExecutionContext<TContext = unknown> =
490
490
  RuntimeContextField<TContext> & {
491
491
  /** The ID of the tool call being executed */
492
492
  toolCallId?: string
493
+ /**
494
+ * Abort signal for the current chat run. Aborts when the run's
495
+ * `abortController` fires (or middleware aborts). Long-running tools —
496
+ * e.g. MCP `callTool` — should forward this to cancel in-flight work.
497
+ */
498
+ abortSignal?: AbortSignal
493
499
  /**
494
500
  * Emit a custom event during tool execution.
495
501
  * Events are streamed to the client in real-time as AG-UI CUSTOM events.
@@ -812,41 +818,6 @@ export interface TextOptions<
812
818
  */
813
819
  systemPrompts?: Array<SystemPrompt>
814
820
  agentLoopStrategy?: AgentLoopStrategy
815
- /**
816
- * Controls the randomness of the output.
817
- * Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
818
- * Range: [0.0, 2.0]
819
- *
820
- * Note: Generally recommended to use either temperature or topP, but not both.
821
- *
822
- * Provider usage:
823
- * - OpenAI: `temperature` (number) - in text.top_p field
824
- * - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
825
- * - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
826
- */
827
- temperature?: number
828
- /**
829
- * Nucleus sampling parameter. An alternative to temperature sampling.
830
- * The model considers the results of tokens with topP probability mass.
831
- * For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
832
- *
833
- * Note: Generally recommended to use either temperature or topP, but not both.
834
- *
835
- * Provider usage:
836
- * - OpenAI: `text.top_p` (number)
837
- * - Anthropic: `top_p` (number | null)
838
- * - Gemini: `generationConfig.topP` (number)
839
- */
840
- topP?: number
841
- /**
842
- * The maximum number of tokens to generate in the response.
843
- *
844
- * Provider usage:
845
- * - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
846
- * - Anthropic: `max_tokens` (number, required) - range x >= 1
847
- * - Gemini: `generationConfig.maxOutputTokens` (number)
848
- */
849
- maxTokens?: number
850
821
  /**
851
822
  * Additional metadata to attach to the request.
852
823
  * Can be used for tracking, debugging, or passing custom information.
@@ -0,0 +1,28 @@
1
+ /**
2
+ * Single source of truth for the provider-native key spellings that cap output
3
+ * tokens. Sampling options live in opaque, provider-native `modelOptions`, and
4
+ * every provider spells the token cap differently. Two call sites must agree on
5
+ * this set or they silently drift:
6
+ *
7
+ * - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied
8
+ * token limit so the summarize default never overrides it.
9
+ * - `middlewares/otel.ts` — picks the first numeric spelling to populate the
10
+ * `gen_ai.request.max_tokens` attribute across providers.
11
+ *
12
+ * Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →
13
+ * native-key map) in the summarize wrapper.
14
+ */
15
+ export const MAX_TOKENS_KEYS = [
16
+ 'max_output_tokens', // OpenAI (Responses)
17
+ 'max_tokens', // Anthropic / Grok
18
+ 'max_completion_tokens', // Groq
19
+ 'maxOutputTokens', // Gemini
20
+ 'maxCompletionTokens', // OpenRouter
21
+ 'maxTokens', // generic / migration leftover (no adapter reads it)
22
+ ] as const
23
+
24
+ /**
25
+ * Ollama nests sampling under `options`; its token cap is `options.num_predict`
26
+ * rather than a flat key.
27
+ */
28
+ export const NESTED_MAX_TOKENS_KEY = 'num_predict' as const