@tanstack/ai 0.26.1 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/dist/esm/activities/chat/index.d.ts +0 -6
  2. package/dist/esm/activities/chat/index.js +2 -17
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/middleware/types.d.ts +1 -4
  5. package/dist/esm/activities/summarize/chat-stream-summarize.js +62 -3
  6. package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
  7. package/dist/esm/logger/internal-logger.d.ts +8 -0
  8. package/dist/esm/logger/internal-logger.js +15 -0
  9. package/dist/esm/logger/internal-logger.js.map +1 -1
  10. package/dist/esm/middlewares/otel.js +30 -6
  11. package/dist/esm/middlewares/otel.js.map +1 -1
  12. package/dist/esm/types.d.ts +0 -35
  13. package/dist/esm/utilities/sampling-keys.d.ts +20 -0
  14. package/dist/esm/utilities/sampling-keys.js +20 -0
  15. package/dist/esm/utilities/sampling-keys.js.map +1 -0
  16. package/package.json +2 -2
  17. package/skills/ai-core/adapter-configuration/SKILL.md +67 -6
  18. package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +6 -3
  19. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +3 -0
  20. package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +10 -1
  21. package/skills/ai-core/adapter-configuration/references/openai-adapter.md +4 -0
  22. package/skills/ai-core/chat-experience/SKILL.md +24 -7
  23. package/skills/ai-core/middleware/SKILL.md +11 -0
  24. package/src/activities/chat/index.ts +2 -23
  25. package/src/activities/chat/middleware/types.ts +1 -4
  26. package/src/activities/summarize/chat-stream-summarize.ts +162 -3
  27. package/src/logger/internal-logger.ts +18 -0
  28. package/src/middlewares/otel.ts +48 -6
  29. package/src/types.ts +0 -35
  30. package/src/utilities/sampling-keys.ts +28 -0
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sampling-keys.js","sources":["../../../src/utilities/sampling-keys.ts"],"sourcesContent":["/**\n * Single source of truth for the provider-native key spellings that cap output\n * tokens. Sampling options live in opaque, provider-native `modelOptions`, and\n * every provider spells the token cap differently. Two call sites must agree on\n * this set or they silently drift:\n *\n * - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied\n * token limit so the summarize default never overrides it.\n * - `middlewares/otel.ts` — picks the first numeric spelling to populate the\n * `gen_ai.request.max_tokens` attribute across providers.\n *\n * Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →\n * native-key map) in the summarize wrapper.\n */\nexport const MAX_TOKENS_KEYS = [\n 'max_output_tokens', // OpenAI (Responses)\n 'max_tokens', // Anthropic / Grok\n 'max_completion_tokens', // Groq\n 'maxOutputTokens', // Gemini\n 'maxCompletionTokens', // OpenRouter\n 'maxTokens', // generic / migration leftover (no adapter reads it)\n] as const\n\n/**\n * Ollama nests sampling under `options`; its token cap is `options.num_predict`\n * rather than a flat key.\n */\nexport const NESTED_MAX_TOKENS_KEY = 'num_predict' as const\n"],"names":[],"mappings":"AAcO,MAAM,kBAAkB;AAAA,EAC7B;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AACF;AAMO,MAAM,wBAAwB;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.26.1",
3
+ "version": "0.27.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -68,7 +68,7 @@
68
68
  "@ag-ui/core": "^0.0.52",
69
69
  "@standard-schema/spec": "^1.1.0",
70
70
  "partial-json": "^0.1.7",
71
- "@tanstack/ai-event-client": "0.5.2"
71
+ "@tanstack/ai-event-client": "0.5.3"
72
72
  },
73
73
  "peerDependencies": {
74
74
  "@opentelemetry/api": ">=1.9.0"
@@ -44,8 +44,10 @@ import { openaiText } from '@tanstack/ai-openai'
44
44
  const stream = chat({
45
45
  adapter: openaiText('gpt-5.2'),
46
46
  messages,
47
- temperature: 0.7,
48
- maxTokens: 1000,
47
+ modelOptions: {
48
+ temperature: 0.7,
49
+ max_output_tokens: 1000,
50
+ },
49
51
  })
50
52
 
51
53
  return toServerSentEventsResponse(stream)
@@ -55,6 +57,11 @@ The adapter factory function takes the model name as a string literal and an
55
57
  optional config object (API key, base URL, etc.). The model name is passed
56
58
  into the factory, not into `chat()`.
57
59
 
60
+ Sampling options (`temperature`, token limits, `top_p`/`topP`, etc.) live
61
+ inside `modelOptions` using each provider's native key — they are **not**
62
+ top-level options on `chat()`. See the per-provider table in
63
+ [Configuring Sampling](#5-configuring-sampling) below.
64
+
58
65
  ## Core Patterns
59
66
 
60
67
  ### 1. Adapter Selection
@@ -158,11 +165,11 @@ const openaiStream = chat({
158
165
  const anthropicStream = chat({
159
166
  adapter: anthropicText('claude-sonnet-4-6'),
160
167
  messages,
161
- maxTokens: 16000,
162
168
  modelOptions: {
169
+ max_tokens: 16000,
163
170
  thinking: {
164
171
  type: 'enabled',
165
- budget_tokens: 8000, // must be >= 1024 and < maxTokens
172
+ budget_tokens: 8000, // must be >= 1024 and < max_tokens
166
173
  },
167
174
  },
168
175
  })
@@ -171,8 +178,8 @@ const anthropicStream = chat({
171
178
  const adaptiveStream = chat({
172
179
  adapter: anthropicText('claude-sonnet-4-6'),
173
180
  messages,
174
- maxTokens: 16000,
175
181
  modelOptions: {
182
+ max_tokens: 16000,
176
183
  thinking: {
177
184
  type: 'adaptive',
178
185
  },
@@ -224,7 +231,61 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
224
231
  At runtime, `extendAdapter` simply passes through to the original factory.
225
232
  The `_customModels` parameter is only used for type inference.
226
233
 
227
- ### 5. Capability Flag: `supportsCombinedToolsAndSchema`
234
+ ### 5. Configuring Sampling
235
+
236
+ Sampling controls (`temperature`, token limits, nucleus sampling) are passed
237
+ inside `modelOptions` using each provider's **native** key. They are not
238
+ top-level fields on `chat()`/`ai()`/`generate()`.
239
+
240
+ ```typescript
241
+ // OpenAI — native keys
242
+ chat({
243
+ adapter: openaiText('gpt-5.2'),
244
+ messages,
245
+ modelOptions: { temperature: 0.7, top_p: 0.9, max_output_tokens: 1000 },
246
+ })
247
+
248
+ // Anthropic
249
+ chat({
250
+ adapter: anthropicText('claude-sonnet-4-6'),
251
+ messages,
252
+ modelOptions: { temperature: 0.7, top_p: 0.9, max_tokens: 1000 },
253
+ })
254
+
255
+ // Gemini — camelCase
256
+ chat({
257
+ adapter: geminiText('gemini-2.5-pro'),
258
+ messages,
259
+ modelOptions: { temperature: 0.7, topP: 0.9, maxOutputTokens: 1000 },
260
+ })
261
+
262
+ // Ollama — NESTED under modelOptions.options
263
+ chat({
264
+ adapter: ollamaText('llama3.3'),
265
+ messages,
266
+ modelOptions: {
267
+ options: { temperature: 0.7, top_p: 0.9, num_predict: 1000 },
268
+ },
269
+ })
270
+ ```
271
+
272
+ Per-provider sampling keys (all live inside `modelOptions`):
273
+
274
+ | Provider | Temperature | Nucleus | Max output tokens |
275
+ | ----------------- | ------------- | ------- | ----------------------------------- |
276
+ | OpenAI | `temperature` | `top_p` | `max_output_tokens` |
277
+ | Anthropic | `temperature` | `top_p` | `max_tokens` |
278
+ | Gemini | `temperature` | `topP` | `maxOutputTokens` |
279
+ | Grok (xAI) | `temperature` | `top_p` | `max_tokens` |
280
+ | Groq | `temperature` | `top_p` | `max_completion_tokens` |
281
+ | OpenRouter (chat) | `temperature` | `topP` | `maxCompletionTokens` |
282
+ | Ollama | `temperature` | `top_p` | `num_predict` (nested in `options`) |
283
+
284
+ `temperature` is the one key every provider names identically; token limits and
285
+ some sampling options use provider-native names. Ollama nests all sampling under
286
+ `modelOptions.options`.
287
+
288
+ ### 6. Capability Flag: `supportsCombinedToolsAndSchema`
228
289
 
229
290
  Adapters can declare an optional capability method:
230
291
 
@@ -39,12 +39,15 @@ Note: Model IDs use the format `claude-opus-4-6`, `claude-sonnet-4-6`, etc.
39
39
  chat({
40
40
  adapter: anthropicText('claude-sonnet-4-6'),
41
41
  messages,
42
- maxTokens: 16000,
43
42
  modelOptions: {
43
+ // Sampling
44
+ temperature: 0.7,
45
+ top_p: 0.9, // cannot be combined with temperature
46
+ max_tokens: 16000,
44
47
  // Extended thinking (budget-based)
45
48
  thinking: {
46
49
  type: 'enabled',
47
- budget_tokens: 8000, // must be >= 1024 and < maxTokens
50
+ budget_tokens: 8000, // must be >= 1024 and < max_tokens
48
51
  },
49
52
  // Adaptive thinking (claude-sonnet-4-6, claude-opus-4-6+)
50
53
  thinking: {
@@ -89,7 +92,7 @@ ANTHROPIC_API_KEY
89
92
 
90
93
  ## Gotchas
91
94
 
92
- - `thinking.budget_tokens` must be >= 1024 AND less than `maxTokens`.
95
+ - `thinking.budget_tokens` must be >= 1024 AND less than `modelOptions.max_tokens`.
93
96
  Failing either check throws a validation error.
94
97
  - Cannot set both `top_p` and `temperature` at the same time (throws error).
95
98
  - `claude-3-5-haiku` and `claude-3-haiku` do NOT support extended thinking.
@@ -72,6 +72,9 @@ chat({
72
72
  // Response modalities
73
73
  responseModalities: ['TEXT'],
74
74
  // Sampling
75
+ temperature: 0.7,
76
+ topP: 0.9,
77
+ maxOutputTokens: 1000,
75
78
  topK: 40,
76
79
  seed: 42,
77
80
  presencePenalty: 0.5,
@@ -40,6 +40,9 @@ Models must be pulled first: `ollama pull llama3.3`
40
40
 
41
41
  Ollama models use a generic options type. Provider options vary by the
42
42
  underlying model. The adapter passes options through to the Ollama API.
43
+ Sampling options are **nested** under `modelOptions.options` (this matches
44
+ Ollama's own request shape) — `temperature`, `top_p`, and `num_predict`
45
+ (max output tokens) all live there.
43
46
 
44
47
  ```typescript
45
48
  import { chat } from '@tanstack/ai'
@@ -48,7 +51,13 @@ import { ollamaText } from '@tanstack/ai-ollama'
48
51
  const stream = chat({
49
52
  adapter: ollamaText('llama3.3'),
50
53
  messages,
51
- temperature: 0.7,
54
+ modelOptions: {
55
+ options: {
56
+ temperature: 0.7,
57
+ top_p: 0.9,
58
+ num_predict: 1000, // max output tokens
59
+ },
60
+ },
52
61
  // Ollama-specific options are limited compared to cloud providers
53
62
  })
54
63
  ```
@@ -44,6 +44,10 @@ chat({
44
44
  adapter: openaiText('gpt-5.4'),
45
45
  messages,
46
46
  modelOptions: {
47
+ // Sampling
48
+ temperature: 0.7,
49
+ top_p: 0.9,
50
+ max_output_tokens: 1000,
47
51
  // Reasoning (effort levels: none, minimal, low, medium, high)
48
52
  reasoning: {
49
53
  effort: 'high',
@@ -137,8 +137,10 @@ import { anthropicText } from '@tanstack/ai-anthropic'
137
137
  const stream = chat({
138
138
  adapter: anthropicText('claude-sonnet-4-5'),
139
139
  messages,
140
- temperature: 0.7,
141
- maxTokens: 2000,
140
+ modelOptions: {
141
+ temperature: 0.7,
142
+ max_tokens: 2000, // Anthropic-native key
143
+ },
142
144
  systemPrompts: ['You are a helpful assistant.'],
143
145
  abortController,
144
146
  })
@@ -377,17 +379,32 @@ chat({ adapter: openaiText('gpt-5.2'), messages })
377
379
 
378
380
  The model is passed to the adapter factory, not to `chat()`.
379
381
 
380
- ### f. HIGH: Nesting temperature/maxTokens in options object
382
+ ### f. HIGH: Passing sampling options at the root of chat()
383
+
384
+ Sampling options (`temperature`, token limits, `top_p`/`topP`) are **not**
385
+ top-level fields on `chat()`. They live inside `modelOptions` using the
386
+ provider's native key.
381
387
 
382
388
  ```typescript
383
- // WRONG
389
+ // WRONG — temperature/maxTokens are not root options
390
+ chat({ adapter, messages, temperature: 0.7, maxTokens: 1000 })
391
+
392
+ // WRONG — there is no `options` field either
384
393
  chat({ adapter, messages, options: { temperature: 0.7, maxTokens: 1000 } })
385
394
 
386
- // CORRECT
387
- chat({ adapter, messages, temperature: 0.7, maxTokens: 1000 })
395
+ // CORRECT — inside modelOptions, provider-native keys (OpenAI shown)
396
+ chat({
397
+ adapter,
398
+ messages,
399
+ modelOptions: { temperature: 0.7, max_output_tokens: 1000 },
400
+ })
388
401
  ```
389
402
 
390
- All parameters are top-level on the `chat()` options object.
403
+ `temperature` is universal across providers; token limits use provider-native
404
+ keys (`max_output_tokens` for OpenAI, `max_tokens` for Anthropic/Grok,
405
+ `maxOutputTokens` for Gemini, `max_completion_tokens` for Groq,
406
+ `maxCompletionTokens` for OpenRouter, and `num_predict` nested under
407
+ `modelOptions.options` for Ollama). See ai-core/adapter-configuration/SKILL.md.
391
408
 
392
409
  ### g. HIGH: Using providerOptions instead of modelOptions
393
410
 
@@ -68,6 +68,13 @@ Every hook receives a `ChatMiddlewareContext` as its first argument, which provi
68
68
  Terminal hooks (`onFinish`, `onAbort`, `onError`) are **mutually exclusive** -- exactly
69
69
  one fires per `chat()` invocation.
70
70
 
71
+ > **Sampling in `onConfig`:** `temperature`, `topP`, and `maxTokens` are **not**
72
+ > first-class fields on `ChatMiddlewareConfig`. To adjust sampling from
73
+ > middleware, return a partial that mutates `config.modelOptions` using the
74
+ > provider's native key (e.g. OpenAI `temperature` / `max_output_tokens`,
75
+ > Anthropic `max_tokens`, Ollama nested `options.num_predict`). Returning a
76
+ > top-level `temperature`/`maxTokens` has no effect.
77
+
71
78
  ### Phase values
72
79
 
73
80
  `ctx.phase` is one of:
@@ -304,6 +311,10 @@ const configTransform: ChatMiddleware = {
304
311
  if (ctx.phase === 'init') {
305
312
  return {
306
313
  systemPrompts: [...config.systemPrompts, 'Always respond in JSON.'],
314
+ // Sampling options are NOT first-class config fields — mutate them
315
+ // through `config.modelOptions` using the provider's native key.
316
+ // (e.g. OpenAI `temperature` / `max_output_tokens`.)
317
+ modelOptions: { ...config.modelOptions, temperature: 0.2 },
307
318
  }
308
319
  }
309
320
  },
@@ -208,12 +208,6 @@ export interface TextActivityOptions<
208
208
  | ProviderTool<string, TAdapter['~types']['toolCapabilities'][number]>
209
209
  >
210
210
  | undefined
211
- /** Controls the randomness of the output. Higher values make output more random. Range: [0.0, 2.0] */
212
- temperature?: TextOptions['temperature']
213
- /** Nucleus sampling parameter. The model considers tokens with topP probability mass. */
214
- topP?: TextOptions['topP']
215
- /** The maximum number of tokens to generate in the response. */
216
- maxTokens?: TextOptions['maxTokens']
217
211
  /** Additional metadata to attach to the request. */
218
212
  metadata?: TextOptions['metadata']
219
213
  /** Model-specific provider options (type comes from adapter) */
@@ -844,13 +838,10 @@ class TextEngine<
844
838
 
845
839
  private beforeRun(): void {
846
840
  this.streamStartTime = Date.now()
847
- const { tools, temperature, topP, maxTokens, metadata } = this.params
841
+ const { tools, metadata } = this.params
848
842
 
849
843
  // Gather flattened options into an object for context
850
844
  const options: Record<string, unknown> = {}
851
- if (temperature !== undefined) options.temperature = temperature
852
- if (topP !== undefined) options.topP = topP
853
- if (maxTokens !== undefined) options.maxTokens = maxTokens
854
845
  if (metadata !== undefined) options.metadata = metadata
855
846
 
856
847
  this.eventOptions = Object.keys(options).length > 0 ? options : undefined
@@ -897,7 +888,7 @@ class TextEngine<
897
888
  }
898
889
 
899
890
  private async *streamModelResponse(): AsyncGenerator<StreamChunk> {
900
- const { temperature, topP, maxTokens, metadata, modelOptions } = this.params
891
+ const { metadata, modelOptions } = this.params
901
892
  const tools = this.tools
902
893
 
903
894
  // Convert tool schemas to JSON Schema before passing to adapter
@@ -941,9 +932,6 @@ class TextEngine<
941
932
  model: this.params.model,
942
933
  messages: this.messages,
943
934
  tools: toolsWithJsonSchemas,
944
- temperature,
945
- topP,
946
- maxTokens,
947
935
  metadata,
948
936
  request: this.effectiveRequest,
949
937
  modelOptions,
@@ -1869,9 +1857,6 @@ class TextEngine<
1869
1857
  chatOptions: {
1870
1858
  model: this.params.model,
1871
1859
  messages: this.messages,
1872
- temperature: postOnConfig.temperature,
1873
- topP: postOnConfig.topP,
1874
- maxTokens: postOnConfig.maxTokens,
1875
1860
  metadata: postOnConfig.metadata,
1876
1861
  modelOptions: postOnConfig.modelOptions,
1877
1862
  systemPrompts: postOnConfig.systemPrompts,
@@ -2351,9 +2336,6 @@ class TextEngine<
2351
2336
  messages: this.messages,
2352
2337
  systemPrompts: [...this.systemPrompts],
2353
2338
  tools: [...this.tools],
2354
- temperature: this.params.temperature,
2355
- topP: this.params.topP,
2356
- maxTokens: this.params.maxTokens,
2357
2339
  metadata: this.params.metadata,
2358
2340
  modelOptions: this.params.modelOptions,
2359
2341
  }
@@ -2365,9 +2347,6 @@ class TextEngine<
2365
2347
  this.tools = config.tools
2366
2348
  this.params = {
2367
2349
  ...this.params,
2368
- temperature: config.temperature,
2369
- topP: config.topP,
2370
- maxTokens: config.maxTokens,
2371
2350
  metadata: config.metadata,
2372
2351
  modelOptions: config.modelOptions,
2373
2352
  }
@@ -90,7 +90,7 @@ export interface ChatMiddlewareContext<TContext = unknown> {
90
90
  systemPrompts: Array<SystemPrompt>
91
91
  /** Names of configured tools, if any */
92
92
  toolNames?: Array<string>
93
- /** Flattened generation options (temperature, topP, maxTokens, metadata) */
93
+ /** Flattened generation options (metadata) */
94
94
  options?: Record<string, unknown> | undefined
95
95
  /** Provider-specific model options */
96
96
  modelOptions?: Record<string, unknown> | undefined
@@ -130,9 +130,6 @@ export interface ChatMiddlewareConfig {
130
130
  messages: Array<ModelMessage>
131
131
  systemPrompts: Array<SystemPrompt>
132
132
  tools: Array<Tool>
133
- temperature?: number
134
- topP?: number
135
- maxTokens?: number
136
133
  metadata?: Record<string, unknown> | undefined
137
134
  modelOptions?: Record<string, unknown> | undefined
138
135
  }
@@ -1,5 +1,6 @@
1
1
  import { EventType } from '@ag-ui/core'
2
2
  import { toRunErrorPayload } from '../error-payload'
3
+ import { MAX_TOKENS_KEYS } from '../../utilities/sampling-keys'
3
4
  import { BaseSummarizeAdapter } from './adapter'
4
5
  import type {
5
6
  StreamChunk,
@@ -23,6 +24,139 @@ export interface ChatStreamCapable {
23
24
  chatStream: (options: TextOptions<any>) => AsyncIterable<StreamChunk>
24
25
  }
25
26
 
27
+ /**
28
+ * Provider-native max-output-tokens key per summarize-adapter `name`. summarize
29
+ * is provider-agnostic and forwards `modelOptions` opaquely to the wrapped text
30
+ * adapter, so `maxLength` must be written under the exact key the underlying
31
+ * provider reads — no adapter reads a generic `maxTokens`. Ollama is the one
32
+ * exception: it nests sampling under `options`, so it has no entry here and is
33
+ * handled as a special nested case in `applyMaxLength`/`applyDefaultTemperature`.
34
+ *
35
+ * Keep in sync with each adapter's wire mapping:
36
+ * - OpenAI (Responses): `max_output_tokens`
37
+ * - Anthropic / Grok: `max_tokens`
38
+ * - Groq: `max_completion_tokens`
39
+ * - Gemini: `maxOutputTokens`
40
+ * - OpenRouter: `maxCompletionTokens`
41
+ * - Ollama: nested `options.num_predict` (no entry — see `applyMaxLength`)
42
+ */
43
+ const MAX_TOKENS_KEY_BY_ADAPTER: Record<string, string> = {
44
+ openai: 'max_output_tokens',
45
+ anthropic: 'max_tokens',
46
+ grok: 'max_tokens',
47
+ groq: 'max_completion_tokens',
48
+ gemini: 'maxOutputTokens',
49
+ openrouter: 'maxCompletionTokens',
50
+ }
51
+
52
+ /**
53
+ * Every flat key any supported provider uses to cap output tokens (plus the
54
+ * generic `maxTokens` spelling no adapter reads). Used to detect a
55
+ * caller-supplied token limit so the summarize default never overrides an
56
+ * explicit caller value. Shared with the OTel middleware via
57
+ * `MAX_TOKENS_KEYS` so the two spelling sets cannot drift.
58
+ */
59
+ const KNOWN_MAX_TOKENS_KEYS = MAX_TOKENS_KEYS
60
+
61
+ /**
62
+ * Whether `applyMaxLength` knows how to place a token limit for this adapter
63
+ * `name` (either the nested Ollama shape or a flat provider-native key).
64
+ * Used to surface a warning when `maxLength` would otherwise be silently
65
+ * dropped for an unrecognised adapter name.
66
+ */
67
+ function isKnownMaxTokensAdapter(adapterName: string): boolean {
68
+ return (
69
+ adapterName === 'ollama' ||
70
+ MAX_TOKENS_KEY_BY_ADAPTER[adapterName] !== undefined
71
+ )
72
+ }
73
+
74
+ /**
75
+ * Apply the low-temperature summarize default to a working copy of the
76
+ * caller's `modelOptions`, placed where the wrapped provider actually reads
77
+ * it (nested under `options` for Ollama, flat otherwise). The caller always
78
+ * wins: if they already set `temperature` in that location, it is untouched.
79
+ */
80
+ function applyDefaultTemperature(
81
+ adapterName: string,
82
+ temperature: number,
83
+ modelOptions: Record<string, unknown>,
84
+ ): Record<string, unknown> {
85
+ const merged: Record<string, unknown> = { ...modelOptions }
86
+
87
+ if (adapterName === 'ollama') {
88
+ const existing =
89
+ merged.options && typeof merged.options === 'object'
90
+ ? (merged.options as Record<string, unknown>)
91
+ : undefined
92
+ if (existing && 'temperature' in existing) return merged
93
+ merged.options = { temperature, ...existing }
94
+ return merged
95
+ }
96
+
97
+ if ('temperature' in merged) return merged
98
+ merged.temperature = temperature
99
+ return merged
100
+ }
101
+
102
+ /**
103
+ * Resolve `maxLength` to the provider-native max-output-tokens key for the
104
+ * given summarize-adapter `name` (this wrapper's OWN `name`, not the wrapped
105
+ * text adapter's) and merge it into a working copy of the caller's
106
+ * `modelOptions`. The caller always wins: if they already set any recognised
107
+ * token-limit key (flat or, for Ollama, nested `options.num_predict`), the
108
+ * default is left untouched. Unknown/unrecognised adapter names fall back to
109
+ * NOT setting a token key (the prompt hint still asks the model to stay under
110
+ * `maxLength`) rather than writing a dead key no provider reads.
111
+ *
112
+ * Caveat (intentional): "caller wins" keys off ANY recognised spelling in
113
+ * `KNOWN_MAX_TOKENS_KEYS`, but only the adapter's native key is read on the
114
+ * wire. So a caller who sets a NON-native spelling for this provider — e.g.
115
+ * `maxTokens`, or Anthropic's `max_tokens` against an OpenAI adapter — suppresses
116
+ * the summarize default WITHOUT getting their own value applied either: neither
117
+ * cap reaches the wire. This favours never clobbering a migration leftover over
118
+ * guaranteeing a cap; the prompt-level hint still asks the model to stay under
119
+ * `maxLength`. Rename the key to the provider-native spelling to forward it.
120
+ */
121
+ function applyMaxLength(
122
+ adapterName: string,
123
+ maxLength: number,
124
+ modelOptions: Record<string, unknown>,
125
+ ): Record<string, unknown> {
126
+ const merged: Record<string, unknown> = { ...modelOptions }
127
+
128
+ if (adapterName === 'ollama') {
129
+ // Honor a caller-set limit in either shape: a recognised flat key (e.g.
130
+ // left over from a migration) or the nested `options.num_predict`.
131
+ const callerSetFlatLimit = KNOWN_MAX_TOKENS_KEYS.some(
132
+ (k) => typeof merged[k] === 'number',
133
+ )
134
+ const existing =
135
+ merged.options && typeof merged.options === 'object'
136
+ ? (merged.options as Record<string, unknown>)
137
+ : undefined
138
+ if (
139
+ callerSetFlatLimit ||
140
+ (existing && typeof existing.num_predict === 'number')
141
+ ) {
142
+ return merged
143
+ }
144
+ merged.options = { num_predict: maxLength, ...existing }
145
+ return merged
146
+ }
147
+
148
+ const key = MAX_TOKENS_KEY_BY_ADAPTER[adapterName]
149
+ if (key === undefined) return merged
150
+
151
+ const callerSetLimit = KNOWN_MAX_TOKENS_KEYS.some(
152
+ (k) => typeof merged[k] === 'number',
153
+ )
154
+ if (callerSetLimit) return merged
155
+
156
+ merged[key] = maxLength
157
+ return merged
158
+ }
159
+
26
160
  /**
27
161
  * Extract the per-model `modelOptions` type a text adapter accepts. Used by
28
162
  * provider summarize factories so their `modelOptions` IntelliSense matches
@@ -195,13 +329,38 @@ export class ChatStreamSummarizeAdapter<
195
329
  options: SummarizationOptions<TProviderOptions>,
196
330
  systemPrompt: string,
197
331
  ): TextOptions<TProviderOptions> {
332
+ // Sampling knobs now live in provider-native `modelOptions`. Apply the
333
+ // low-temperature default where the wrapped provider actually reads it
334
+ // (nested under `options` for Ollama, flat otherwise) so callers can still
335
+ // override it. Resolving the placement from this summarize adapter's OWN
336
+ // `name` keeps the default off the wire correctly per provider — a flat
337
+ // `temperature` would be silently dropped by Ollama while still showing up
338
+ // in OTel.
339
+ let working: Record<string, unknown> = {
340
+ ...(options.modelOptions as Record<string, unknown> | undefined),
341
+ }
342
+ working = applyDefaultTemperature(this.name, 0.3, working)
343
+ // `maxLength` must reach the wire under the provider-native token key (it
344
+ // differs per provider, and no adapter reads a generic `maxTokens`).
345
+ // Resolve it from this summarize adapter's `name` (the constructor arg,
346
+ // not the wrapped text adapter's name), never overriding a caller-supplied
347
+ // token limit.
348
+ if (options.maxLength !== undefined) {
349
+ if (!isKnownMaxTokensAdapter(this.name)) {
350
+ options.logger.warn(
351
+ `summarize: maxLength=${options.maxLength} could not be mapped to a provider token key for adapter name "${this.name}" — it was dropped from modelOptions (the prompt still asks the model to stay under it). Construct ChatStreamSummarizeAdapter with a recognised provider name to forward the cap.`,
352
+ { provider: this.name },
353
+ )
354
+ }
355
+ working = applyMaxLength(this.name, options.maxLength, working)
356
+ }
357
+ const modelOptions = working as TProviderOptions
358
+
198
359
  return {
199
360
  model: options.model,
200
361
  messages: [{ role: 'user', content: options.text }],
201
362
  systemPrompts: [systemPrompt],
202
- maxTokens: options.maxLength,
203
- temperature: 0.3,
204
- modelOptions: options.modelOptions,
363
+ modelOptions,
205
364
  logger: options.logger,
206
365
  }
207
366
  }
@@ -104,4 +104,22 @@ export class InternalLogger {
104
104
  request(message: string, meta?: Record<string, unknown>): void {
105
105
  this.emit('debug', 'request', message, meta)
106
106
  }
107
+
108
+ /**
109
+ * Log a non-fatal misconfiguration or recoverable anomaly. Gated by the
110
+ * `errors` category — on by default (and when `debug` is unspecified), so
111
+ * silent-drop conditions surface, but still silenced by `debug: false`,
112
+ * which honors the "disable everything including errors" contract. Routes to
113
+ * the underlying logger's `warn` level.
114
+ */
115
+ warn(message: string, meta?: Record<string, unknown>): void {
116
+ if (!this.categories.errors) return
117
+ const prefixed = `⚠️ [tanstack-ai:warn] ⚠️ ${message}`
118
+ try {
119
+ this.logger.warn(prefixed, meta)
120
+ } catch {
121
+ // User-supplied logger threw; swallow so a broken logger never masks the
122
+ // condition we were trying to surface.
123
+ }
124
+ }
107
125
  }
@@ -4,6 +4,10 @@ import {
4
4
  context as otelContext,
5
5
  trace as otelTrace,
6
6
  } from '@opentelemetry/api'
7
+ import {
8
+ MAX_TOKENS_KEYS,
9
+ NESTED_MAX_TOKENS_KEY,
10
+ } from '../utilities/sampling-keys'
7
11
  import type {
8
12
  AttributeValue,
9
13
  Exception,
@@ -162,6 +166,19 @@ function messageEventName(role: string): string {
162
166
  }
163
167
  }
164
168
 
169
+ /**
170
+ * Return the first candidate that is a finite `number`, or `undefined`. Used to
171
+ * pick a sampling attribute from among the several provider-native spellings.
172
+ */
173
+ function firstNumber(...candidates: Array<unknown>): number | undefined {
174
+ for (const candidate of candidates) {
175
+ if (typeof candidate === 'number' && Number.isFinite(candidate)) {
176
+ return candidate
177
+ }
178
+ }
179
+ return undefined
180
+ }
181
+
165
182
  function errorMessage(err: unknown): string | undefined {
166
183
  if (err instanceof Error) return err.message
167
184
  if (typeof err === 'string') return err
@@ -333,12 +350,37 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
333
350
  'gen_ai.request.model': ctx.model,
334
351
  'tanstack.ai.iteration': ctx.iteration,
335
352
  }
336
- if (config.temperature !== undefined)
337
- baseAttrs['gen_ai.request.temperature'] = config.temperature
338
- if (config.topP !== undefined)
339
- baseAttrs['gen_ai.request.top_p'] = config.topP
340
- if (config.maxTokens !== undefined)
341
- baseAttrs['gen_ai.request.max_tokens'] = config.maxTokens
353
+ // Sampling options now live in provider-native `modelOptions`, and
354
+ // providers spell them differently (e.g. `max_output_tokens`,
355
+ // `max_completion_tokens`, `maxOutputTokens`, `num_predict`). Read the
356
+ // first numeric value among the known spellings — including Ollama's
357
+ // nested `options` — so gen_ai attributes populate across providers.
358
+ const sampling = config.modelOptions ?? {}
359
+ const nestedOptions =
360
+ sampling['options'] && typeof sampling['options'] === 'object'
361
+ ? (sampling['options'] as Record<string, unknown>)
362
+ : undefined
363
+ const samplingTemperature = firstNumber(
364
+ sampling['temperature'],
365
+ nestedOptions?.['temperature'],
366
+ )
367
+ const samplingTopP = firstNumber(
368
+ sampling['top_p'],
369
+ sampling['topP'],
370
+ nestedOptions?.['top_p'],
371
+ )
372
+ // Spellings come from the shared `MAX_TOKENS_KEYS` table so this stays
373
+ // in lockstep with the summarize wrapper's caller-limit detection.
374
+ const samplingMaxTokens = firstNumber(
375
+ ...MAX_TOKENS_KEYS.map((k) => sampling[k]),
376
+ nestedOptions?.[NESTED_MAX_TOKENS_KEY],
377
+ )
378
+ if (samplingTemperature !== undefined)
379
+ baseAttrs['gen_ai.request.temperature'] = samplingTemperature
380
+ if (samplingTopP !== undefined)
381
+ baseAttrs['gen_ai.request.top_p'] = samplingTopP
382
+ if (samplingMaxTokens !== undefined)
383
+ baseAttrs['gen_ai.request.max_tokens'] = samplingMaxTokens
342
384
 
343
385
  const baseOptions: SpanOptions = {
344
386
  kind: SpanKind.CLIENT,