@tanstack/ai 0.26.1 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/index.d.ts +0 -6
- package/dist/esm/activities/chat/index.js +2 -17
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +1 -4
- package/dist/esm/activities/summarize/chat-stream-summarize.js +62 -3
- package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
- package/dist/esm/logger/internal-logger.d.ts +8 -0
- package/dist/esm/logger/internal-logger.js +15 -0
- package/dist/esm/logger/internal-logger.js.map +1 -1
- package/dist/esm/middlewares/otel.js +30 -6
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/types.d.ts +0 -35
- package/dist/esm/utilities/sampling-keys.d.ts +20 -0
- package/dist/esm/utilities/sampling-keys.js +20 -0
- package/dist/esm/utilities/sampling-keys.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +67 -6
- package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +6 -3
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +3 -0
- package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +10 -1
- package/skills/ai-core/adapter-configuration/references/openai-adapter.md +4 -0
- package/skills/ai-core/chat-experience/SKILL.md +24 -7
- package/skills/ai-core/middleware/SKILL.md +11 -0
- package/src/activities/chat/index.ts +2 -23
- package/src/activities/chat/middleware/types.ts +1 -4
- package/src/activities/summarize/chat-stream-summarize.ts +162 -3
- package/src/logger/internal-logger.ts +18 -0
- package/src/middlewares/otel.ts +48 -6
- package/src/types.ts +0 -35
- package/src/utilities/sampling-keys.ts +28 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sampling-keys.js","sources":["../../../src/utilities/sampling-keys.ts"],"sourcesContent":["/**\n * Single source of truth for the provider-native key spellings that cap output\n * tokens. Sampling options live in opaque, provider-native `modelOptions`, and\n * every provider spells the token cap differently. Two call sites must agree on\n * this set or they silently drift:\n *\n * - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied\n * token limit so the summarize default never overrides it.\n * - `middlewares/otel.ts` — picks the first numeric spelling to populate the\n * `gen_ai.request.max_tokens` attribute across providers.\n *\n * Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →\n * native-key map) in the summarize wrapper.\n */\nexport const MAX_TOKENS_KEYS = [\n 'max_output_tokens', // OpenAI (Responses)\n 'max_tokens', // Anthropic / Grok\n 'max_completion_tokens', // Groq\n 'maxOutputTokens', // Gemini\n 'maxCompletionTokens', // OpenRouter\n 'maxTokens', // generic / migration leftover (no adapter reads it)\n] as const\n\n/**\n * Ollama nests sampling under `options`; its token cap is `options.num_predict`\n * rather than a flat key.\n */\nexport const NESTED_MAX_TOKENS_KEY = 'num_predict' as const\n"],"names":[],"mappings":"AAcO,MAAM,kBAAkB;AAAA,EAC7B;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AACF;AAMO,MAAM,wBAAwB;"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.27.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -68,7 +68,7 @@
|
|
|
68
68
|
"@ag-ui/core": "^0.0.52",
|
|
69
69
|
"@standard-schema/spec": "^1.1.0",
|
|
70
70
|
"partial-json": "^0.1.7",
|
|
71
|
-
"@tanstack/ai-event-client": "0.5.
|
|
71
|
+
"@tanstack/ai-event-client": "0.5.3"
|
|
72
72
|
},
|
|
73
73
|
"peerDependencies": {
|
|
74
74
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -44,8 +44,10 @@ import { openaiText } from '@tanstack/ai-openai'
|
|
|
44
44
|
const stream = chat({
|
|
45
45
|
adapter: openaiText('gpt-5.2'),
|
|
46
46
|
messages,
|
|
47
|
-
|
|
48
|
-
|
|
47
|
+
modelOptions: {
|
|
48
|
+
temperature: 0.7,
|
|
49
|
+
max_output_tokens: 1000,
|
|
50
|
+
},
|
|
49
51
|
})
|
|
50
52
|
|
|
51
53
|
return toServerSentEventsResponse(stream)
|
|
@@ -55,6 +57,11 @@ The adapter factory function takes the model name as a string literal and an
|
|
|
55
57
|
optional config object (API key, base URL, etc.). The model name is passed
|
|
56
58
|
into the factory, not into `chat()`.
|
|
57
59
|
|
|
60
|
+
Sampling options (`temperature`, token limits, `top_p`/`topP`, etc.) live
|
|
61
|
+
inside `modelOptions` using each provider's native key — they are **not**
|
|
62
|
+
top-level options on `chat()`. See the per-provider table in
|
|
63
|
+
[Configuring Sampling](#5-configuring-sampling) below.
|
|
64
|
+
|
|
58
65
|
## Core Patterns
|
|
59
66
|
|
|
60
67
|
### 1. Adapter Selection
|
|
@@ -158,11 +165,11 @@ const openaiStream = chat({
|
|
|
158
165
|
const anthropicStream = chat({
|
|
159
166
|
adapter: anthropicText('claude-sonnet-4-6'),
|
|
160
167
|
messages,
|
|
161
|
-
maxTokens: 16000,
|
|
162
168
|
modelOptions: {
|
|
169
|
+
max_tokens: 16000,
|
|
163
170
|
thinking: {
|
|
164
171
|
type: 'enabled',
|
|
165
|
-
budget_tokens: 8000, // must be >= 1024 and <
|
|
172
|
+
budget_tokens: 8000, // must be >= 1024 and < max_tokens
|
|
166
173
|
},
|
|
167
174
|
},
|
|
168
175
|
})
|
|
@@ -171,8 +178,8 @@ const anthropicStream = chat({
|
|
|
171
178
|
const adaptiveStream = chat({
|
|
172
179
|
adapter: anthropicText('claude-sonnet-4-6'),
|
|
173
180
|
messages,
|
|
174
|
-
maxTokens: 16000,
|
|
175
181
|
modelOptions: {
|
|
182
|
+
max_tokens: 16000,
|
|
176
183
|
thinking: {
|
|
177
184
|
type: 'adaptive',
|
|
178
185
|
},
|
|
@@ -224,7 +231,61 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
|
|
|
224
231
|
At runtime, `extendAdapter` simply passes through to the original factory.
|
|
225
232
|
The `_customModels` parameter is only used for type inference.
|
|
226
233
|
|
|
227
|
-
### 5.
|
|
234
|
+
### 5. Configuring Sampling
|
|
235
|
+
|
|
236
|
+
Sampling controls (`temperature`, token limits, nucleus sampling) are passed
|
|
237
|
+
inside `modelOptions` using each provider's **native** key. They are not
|
|
238
|
+
top-level fields on `chat()`/`ai()`/`generate()`.
|
|
239
|
+
|
|
240
|
+
```typescript
|
|
241
|
+
// OpenAI — native keys
|
|
242
|
+
chat({
|
|
243
|
+
adapter: openaiText('gpt-5.2'),
|
|
244
|
+
messages,
|
|
245
|
+
modelOptions: { temperature: 0.7, top_p: 0.9, max_output_tokens: 1000 },
|
|
246
|
+
})
|
|
247
|
+
|
|
248
|
+
// Anthropic
|
|
249
|
+
chat({
|
|
250
|
+
adapter: anthropicText('claude-sonnet-4-6'),
|
|
251
|
+
messages,
|
|
252
|
+
modelOptions: { temperature: 0.7, top_p: 0.9, max_tokens: 1000 },
|
|
253
|
+
})
|
|
254
|
+
|
|
255
|
+
// Gemini — camelCase
|
|
256
|
+
chat({
|
|
257
|
+
adapter: geminiText('gemini-2.5-pro'),
|
|
258
|
+
messages,
|
|
259
|
+
modelOptions: { temperature: 0.7, topP: 0.9, maxOutputTokens: 1000 },
|
|
260
|
+
})
|
|
261
|
+
|
|
262
|
+
// Ollama — NESTED under modelOptions.options
|
|
263
|
+
chat({
|
|
264
|
+
adapter: ollamaText('llama3.3'),
|
|
265
|
+
messages,
|
|
266
|
+
modelOptions: {
|
|
267
|
+
options: { temperature: 0.7, top_p: 0.9, num_predict: 1000 },
|
|
268
|
+
},
|
|
269
|
+
})
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
Per-provider sampling keys (all live inside `modelOptions`):
|
|
273
|
+
|
|
274
|
+
| Provider | Temperature | Nucleus | Max output tokens |
|
|
275
|
+
| ----------------- | ------------- | ------- | ----------------------------------- |
|
|
276
|
+
| OpenAI | `temperature` | `top_p` | `max_output_tokens` |
|
|
277
|
+
| Anthropic | `temperature` | `top_p` | `max_tokens` |
|
|
278
|
+
| Gemini | `temperature` | `topP` | `maxOutputTokens` |
|
|
279
|
+
| Grok (xAI) | `temperature` | `top_p` | `max_tokens` |
|
|
280
|
+
| Groq | `temperature` | `top_p` | `max_completion_tokens` |
|
|
281
|
+
| OpenRouter (chat) | `temperature` | `topP` | `maxCompletionTokens` |
|
|
282
|
+
| Ollama | `temperature` | `top_p` | `num_predict` (nested in `options`) |
|
|
283
|
+
|
|
284
|
+
`temperature` is the one key every provider names identically; token limits and
|
|
285
|
+
some sampling options use provider-native names. Ollama nests all sampling under
|
|
286
|
+
`modelOptions.options`.
|
|
287
|
+
|
|
288
|
+
### 6. Capability Flag: `supportsCombinedToolsAndSchema`
|
|
228
289
|
|
|
229
290
|
Adapters can declare an optional capability method:
|
|
230
291
|
|
|
@@ -39,12 +39,15 @@ Note: Model IDs use the format `claude-opus-4-6`, `claude-sonnet-4-6`, etc.
|
|
|
39
39
|
chat({
|
|
40
40
|
adapter: anthropicText('claude-sonnet-4-6'),
|
|
41
41
|
messages,
|
|
42
|
-
maxTokens: 16000,
|
|
43
42
|
modelOptions: {
|
|
43
|
+
// Sampling
|
|
44
|
+
temperature: 0.7,
|
|
45
|
+
top_p: 0.9, // cannot be combined with temperature
|
|
46
|
+
max_tokens: 16000,
|
|
44
47
|
// Extended thinking (budget-based)
|
|
45
48
|
thinking: {
|
|
46
49
|
type: 'enabled',
|
|
47
|
-
budget_tokens: 8000, // must be >= 1024 and <
|
|
50
|
+
budget_tokens: 8000, // must be >= 1024 and < max_tokens
|
|
48
51
|
},
|
|
49
52
|
// Adaptive thinking (claude-sonnet-4-6, claude-opus-4-6+)
|
|
50
53
|
thinking: {
|
|
@@ -89,7 +92,7 @@ ANTHROPIC_API_KEY
|
|
|
89
92
|
|
|
90
93
|
## Gotchas
|
|
91
94
|
|
|
92
|
-
- `thinking.budget_tokens` must be >= 1024 AND less than `
|
|
95
|
+
- `thinking.budget_tokens` must be >= 1024 AND less than `modelOptions.max_tokens`.
|
|
93
96
|
Failing either check throws a validation error.
|
|
94
97
|
- Cannot set both `top_p` and `temperature` at the same time (throws error).
|
|
95
98
|
- `claude-3-5-haiku` and `claude-3-haiku` do NOT support extended thinking.
|
|
@@ -40,6 +40,9 @@ Models must be pulled first: `ollama pull llama3.3`
|
|
|
40
40
|
|
|
41
41
|
Ollama models use a generic options type. Provider options vary by the
|
|
42
42
|
underlying model. The adapter passes options through to the Ollama API.
|
|
43
|
+
Sampling options are **nested** under `modelOptions.options` (this matches
|
|
44
|
+
Ollama's own request shape) — `temperature`, `top_p`, and `num_predict`
|
|
45
|
+
(max output tokens) all live there.
|
|
43
46
|
|
|
44
47
|
```typescript
|
|
45
48
|
import { chat } from '@tanstack/ai'
|
|
@@ -48,7 +51,13 @@ import { ollamaText } from '@tanstack/ai-ollama'
|
|
|
48
51
|
const stream = chat({
|
|
49
52
|
adapter: ollamaText('llama3.3'),
|
|
50
53
|
messages,
|
|
51
|
-
|
|
54
|
+
modelOptions: {
|
|
55
|
+
options: {
|
|
56
|
+
temperature: 0.7,
|
|
57
|
+
top_p: 0.9,
|
|
58
|
+
num_predict: 1000, // max output tokens
|
|
59
|
+
},
|
|
60
|
+
},
|
|
52
61
|
// Ollama-specific options are limited compared to cloud providers
|
|
53
62
|
})
|
|
54
63
|
```
|
|
@@ -137,8 +137,10 @@ import { anthropicText } from '@tanstack/ai-anthropic'
|
|
|
137
137
|
const stream = chat({
|
|
138
138
|
adapter: anthropicText('claude-sonnet-4-5'),
|
|
139
139
|
messages,
|
|
140
|
-
|
|
141
|
-
|
|
140
|
+
modelOptions: {
|
|
141
|
+
temperature: 0.7,
|
|
142
|
+
max_tokens: 2000, // Anthropic-native key
|
|
143
|
+
},
|
|
142
144
|
systemPrompts: ['You are a helpful assistant.'],
|
|
143
145
|
abortController,
|
|
144
146
|
})
|
|
@@ -377,17 +379,32 @@ chat({ adapter: openaiText('gpt-5.2'), messages })
|
|
|
377
379
|
|
|
378
380
|
The model is passed to the adapter factory, not to `chat()`.
|
|
379
381
|
|
|
380
|
-
### f. HIGH:
|
|
382
|
+
### f. HIGH: Passing sampling options at the root of chat()
|
|
383
|
+
|
|
384
|
+
Sampling options (`temperature`, token limits, `top_p`/`topP`) are **not**
|
|
385
|
+
top-level fields on `chat()`. They live inside `modelOptions` using the
|
|
386
|
+
provider's native key.
|
|
381
387
|
|
|
382
388
|
```typescript
|
|
383
|
-
// WRONG
|
|
389
|
+
// WRONG — temperature/maxTokens are not root options
|
|
390
|
+
chat({ adapter, messages, temperature: 0.7, maxTokens: 1000 })
|
|
391
|
+
|
|
392
|
+
// WRONG — there is no `options` field either
|
|
384
393
|
chat({ adapter, messages, options: { temperature: 0.7, maxTokens: 1000 } })
|
|
385
394
|
|
|
386
|
-
// CORRECT
|
|
387
|
-
chat({
|
|
395
|
+
// CORRECT — inside modelOptions, provider-native keys (OpenAI shown)
|
|
396
|
+
chat({
|
|
397
|
+
adapter,
|
|
398
|
+
messages,
|
|
399
|
+
modelOptions: { temperature: 0.7, max_output_tokens: 1000 },
|
|
400
|
+
})
|
|
388
401
|
```
|
|
389
402
|
|
|
390
|
-
|
|
403
|
+
`temperature` is universal across providers; token limits use provider-native
|
|
404
|
+
keys (`max_output_tokens` for OpenAI, `max_tokens` for Anthropic/Grok,
|
|
405
|
+
`maxOutputTokens` for Gemini, `max_completion_tokens` for Groq,
|
|
406
|
+
`maxCompletionTokens` for OpenRouter, and `num_predict` nested under
|
|
407
|
+
`modelOptions.options` for Ollama). See ai-core/adapter-configuration/SKILL.md.
|
|
391
408
|
|
|
392
409
|
### g. HIGH: Using providerOptions instead of modelOptions
|
|
393
410
|
|
|
@@ -68,6 +68,13 @@ Every hook receives a `ChatMiddlewareContext` as its first argument, which provi
|
|
|
68
68
|
Terminal hooks (`onFinish`, `onAbort`, `onError`) are **mutually exclusive** -- exactly
|
|
69
69
|
one fires per `chat()` invocation.
|
|
70
70
|
|
|
71
|
+
> **Sampling in `onConfig`:** `temperature`, `topP`, and `maxTokens` are **not**
|
|
72
|
+
> first-class fields on `ChatMiddlewareConfig`. To adjust sampling from
|
|
73
|
+
> middleware, return a partial that mutates `config.modelOptions` using the
|
|
74
|
+
> provider's native key (e.g. OpenAI `temperature` / `max_output_tokens`,
|
|
75
|
+
> Anthropic `max_tokens`, Ollama nested `options.num_predict`). Returning a
|
|
76
|
+
> top-level `temperature`/`maxTokens` has no effect.
|
|
77
|
+
|
|
71
78
|
### Phase values
|
|
72
79
|
|
|
73
80
|
`ctx.phase` is one of:
|
|
@@ -304,6 +311,10 @@ const configTransform: ChatMiddleware = {
|
|
|
304
311
|
if (ctx.phase === 'init') {
|
|
305
312
|
return {
|
|
306
313
|
systemPrompts: [...config.systemPrompts, 'Always respond in JSON.'],
|
|
314
|
+
// Sampling options are NOT first-class config fields — mutate them
|
|
315
|
+
// through `config.modelOptions` using the provider's native key.
|
|
316
|
+
// (e.g. OpenAI `temperature` / `max_output_tokens`.)
|
|
317
|
+
modelOptions: { ...config.modelOptions, temperature: 0.2 },
|
|
307
318
|
}
|
|
308
319
|
}
|
|
309
320
|
},
|
|
@@ -208,12 +208,6 @@ export interface TextActivityOptions<
|
|
|
208
208
|
| ProviderTool<string, TAdapter['~types']['toolCapabilities'][number]>
|
|
209
209
|
>
|
|
210
210
|
| undefined
|
|
211
|
-
/** Controls the randomness of the output. Higher values make output more random. Range: [0.0, 2.0] */
|
|
212
|
-
temperature?: TextOptions['temperature']
|
|
213
|
-
/** Nucleus sampling parameter. The model considers tokens with topP probability mass. */
|
|
214
|
-
topP?: TextOptions['topP']
|
|
215
|
-
/** The maximum number of tokens to generate in the response. */
|
|
216
|
-
maxTokens?: TextOptions['maxTokens']
|
|
217
211
|
/** Additional metadata to attach to the request. */
|
|
218
212
|
metadata?: TextOptions['metadata']
|
|
219
213
|
/** Model-specific provider options (type comes from adapter) */
|
|
@@ -844,13 +838,10 @@ class TextEngine<
|
|
|
844
838
|
|
|
845
839
|
private beforeRun(): void {
|
|
846
840
|
this.streamStartTime = Date.now()
|
|
847
|
-
const { tools,
|
|
841
|
+
const { tools, metadata } = this.params
|
|
848
842
|
|
|
849
843
|
// Gather flattened options into an object for context
|
|
850
844
|
const options: Record<string, unknown> = {}
|
|
851
|
-
if (temperature !== undefined) options.temperature = temperature
|
|
852
|
-
if (topP !== undefined) options.topP = topP
|
|
853
|
-
if (maxTokens !== undefined) options.maxTokens = maxTokens
|
|
854
845
|
if (metadata !== undefined) options.metadata = metadata
|
|
855
846
|
|
|
856
847
|
this.eventOptions = Object.keys(options).length > 0 ? options : undefined
|
|
@@ -897,7 +888,7 @@ class TextEngine<
|
|
|
897
888
|
}
|
|
898
889
|
|
|
899
890
|
private async *streamModelResponse(): AsyncGenerator<StreamChunk> {
|
|
900
|
-
const {
|
|
891
|
+
const { metadata, modelOptions } = this.params
|
|
901
892
|
const tools = this.tools
|
|
902
893
|
|
|
903
894
|
// Convert tool schemas to JSON Schema before passing to adapter
|
|
@@ -941,9 +932,6 @@ class TextEngine<
|
|
|
941
932
|
model: this.params.model,
|
|
942
933
|
messages: this.messages,
|
|
943
934
|
tools: toolsWithJsonSchemas,
|
|
944
|
-
temperature,
|
|
945
|
-
topP,
|
|
946
|
-
maxTokens,
|
|
947
935
|
metadata,
|
|
948
936
|
request: this.effectiveRequest,
|
|
949
937
|
modelOptions,
|
|
@@ -1869,9 +1857,6 @@ class TextEngine<
|
|
|
1869
1857
|
chatOptions: {
|
|
1870
1858
|
model: this.params.model,
|
|
1871
1859
|
messages: this.messages,
|
|
1872
|
-
temperature: postOnConfig.temperature,
|
|
1873
|
-
topP: postOnConfig.topP,
|
|
1874
|
-
maxTokens: postOnConfig.maxTokens,
|
|
1875
1860
|
metadata: postOnConfig.metadata,
|
|
1876
1861
|
modelOptions: postOnConfig.modelOptions,
|
|
1877
1862
|
systemPrompts: postOnConfig.systemPrompts,
|
|
@@ -2351,9 +2336,6 @@ class TextEngine<
|
|
|
2351
2336
|
messages: this.messages,
|
|
2352
2337
|
systemPrompts: [...this.systemPrompts],
|
|
2353
2338
|
tools: [...this.tools],
|
|
2354
|
-
temperature: this.params.temperature,
|
|
2355
|
-
topP: this.params.topP,
|
|
2356
|
-
maxTokens: this.params.maxTokens,
|
|
2357
2339
|
metadata: this.params.metadata,
|
|
2358
2340
|
modelOptions: this.params.modelOptions,
|
|
2359
2341
|
}
|
|
@@ -2365,9 +2347,6 @@ class TextEngine<
|
|
|
2365
2347
|
this.tools = config.tools
|
|
2366
2348
|
this.params = {
|
|
2367
2349
|
...this.params,
|
|
2368
|
-
temperature: config.temperature,
|
|
2369
|
-
topP: config.topP,
|
|
2370
|
-
maxTokens: config.maxTokens,
|
|
2371
2350
|
metadata: config.metadata,
|
|
2372
2351
|
modelOptions: config.modelOptions,
|
|
2373
2352
|
}
|
|
@@ -90,7 +90,7 @@ export interface ChatMiddlewareContext<TContext = unknown> {
|
|
|
90
90
|
systemPrompts: Array<SystemPrompt>
|
|
91
91
|
/** Names of configured tools, if any */
|
|
92
92
|
toolNames?: Array<string>
|
|
93
|
-
/** Flattened generation options (
|
|
93
|
+
/** Flattened generation options (metadata) */
|
|
94
94
|
options?: Record<string, unknown> | undefined
|
|
95
95
|
/** Provider-specific model options */
|
|
96
96
|
modelOptions?: Record<string, unknown> | undefined
|
|
@@ -130,9 +130,6 @@ export interface ChatMiddlewareConfig {
|
|
|
130
130
|
messages: Array<ModelMessage>
|
|
131
131
|
systemPrompts: Array<SystemPrompt>
|
|
132
132
|
tools: Array<Tool>
|
|
133
|
-
temperature?: number
|
|
134
|
-
topP?: number
|
|
135
|
-
maxTokens?: number
|
|
136
133
|
metadata?: Record<string, unknown> | undefined
|
|
137
134
|
modelOptions?: Record<string, unknown> | undefined
|
|
138
135
|
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { EventType } from '@ag-ui/core'
|
|
2
2
|
import { toRunErrorPayload } from '../error-payload'
|
|
3
|
+
import { MAX_TOKENS_KEYS } from '../../utilities/sampling-keys'
|
|
3
4
|
import { BaseSummarizeAdapter } from './adapter'
|
|
4
5
|
import type {
|
|
5
6
|
StreamChunk,
|
|
@@ -23,6 +24,139 @@ export interface ChatStreamCapable {
|
|
|
23
24
|
chatStream: (options: TextOptions<any>) => AsyncIterable<StreamChunk>
|
|
24
25
|
}
|
|
25
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Provider-native max-output-tokens key per summarize-adapter `name`. summarize
|
|
29
|
+
* is provider-agnostic and forwards `modelOptions` opaquely to the wrapped text
|
|
30
|
+
* adapter, so `maxLength` must be written under the exact key the underlying
|
|
31
|
+
* provider reads — no adapter reads a generic `maxTokens`. Ollama is the one
|
|
32
|
+
* exception: it nests sampling under `options`, so it has no entry here and is
|
|
33
|
+
* handled as a special nested case in `applyMaxLength`/`applyDefaultTemperature`.
|
|
34
|
+
*
|
|
35
|
+
* Keep in sync with each adapter's wire mapping:
|
|
36
|
+
* - OpenAI (Responses): `max_output_tokens`
|
|
37
|
+
* - Anthropic / Grok: `max_tokens`
|
|
38
|
+
* - Groq: `max_completion_tokens`
|
|
39
|
+
* - Gemini: `maxOutputTokens`
|
|
40
|
+
* - OpenRouter: `maxCompletionTokens`
|
|
41
|
+
* - Ollama: nested `options.num_predict` (no entry — see `applyMaxLength`)
|
|
42
|
+
*/
|
|
43
|
+
const MAX_TOKENS_KEY_BY_ADAPTER: Record<string, string> = {
|
|
44
|
+
openai: 'max_output_tokens',
|
|
45
|
+
anthropic: 'max_tokens',
|
|
46
|
+
grok: 'max_tokens',
|
|
47
|
+
groq: 'max_completion_tokens',
|
|
48
|
+
gemini: 'maxOutputTokens',
|
|
49
|
+
openrouter: 'maxCompletionTokens',
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Every flat key any supported provider uses to cap output tokens (plus the
|
|
54
|
+
* generic `maxTokens` spelling no adapter reads). Used to detect a
|
|
55
|
+
* caller-supplied token limit so the summarize default never overrides an
|
|
56
|
+
* explicit caller value. Shared with the OTel middleware via
|
|
57
|
+
* `MAX_TOKENS_KEYS` so the two spelling sets cannot drift.
|
|
58
|
+
*/
|
|
59
|
+
const KNOWN_MAX_TOKENS_KEYS = MAX_TOKENS_KEYS
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Whether `applyMaxLength` knows how to place a token limit for this adapter
|
|
63
|
+
* `name` (either the nested Ollama shape or a flat provider-native key).
|
|
64
|
+
* Used to surface a warning when `maxLength` would otherwise be silently
|
|
65
|
+
* dropped for an unrecognised adapter name.
|
|
66
|
+
*/
|
|
67
|
+
function isKnownMaxTokensAdapter(adapterName: string): boolean {
|
|
68
|
+
return (
|
|
69
|
+
adapterName === 'ollama' ||
|
|
70
|
+
MAX_TOKENS_KEY_BY_ADAPTER[adapterName] !== undefined
|
|
71
|
+
)
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Apply the low-temperature summarize default to a working copy of the
|
|
76
|
+
* caller's `modelOptions`, placed where the wrapped provider actually reads
|
|
77
|
+
* it (nested under `options` for Ollama, flat otherwise). The caller always
|
|
78
|
+
* wins: if they already set `temperature` in that location, it is untouched.
|
|
79
|
+
*/
|
|
80
|
+
function applyDefaultTemperature(
|
|
81
|
+
adapterName: string,
|
|
82
|
+
temperature: number,
|
|
83
|
+
modelOptions: Record<string, unknown>,
|
|
84
|
+
): Record<string, unknown> {
|
|
85
|
+
const merged: Record<string, unknown> = { ...modelOptions }
|
|
86
|
+
|
|
87
|
+
if (adapterName === 'ollama') {
|
|
88
|
+
const existing =
|
|
89
|
+
merged.options && typeof merged.options === 'object'
|
|
90
|
+
? (merged.options as Record<string, unknown>)
|
|
91
|
+
: undefined
|
|
92
|
+
if (existing && 'temperature' in existing) return merged
|
|
93
|
+
merged.options = { temperature, ...existing }
|
|
94
|
+
return merged
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
if ('temperature' in merged) return merged
|
|
98
|
+
merged.temperature = temperature
|
|
99
|
+
return merged
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Resolve `maxLength` to the provider-native max-output-tokens key for the
|
|
104
|
+
* given summarize-adapter `name` (this wrapper's OWN `name`, not the wrapped
|
|
105
|
+
* text adapter's) and merge it into a working copy of the caller's
|
|
106
|
+
* `modelOptions`. The caller always wins: if they already set any recognised
|
|
107
|
+
* token-limit key (flat or, for Ollama, nested `options.num_predict`), the
|
|
108
|
+
* default is left untouched. Unknown/unrecognised adapter names fall back to
|
|
109
|
+
* NOT setting a token key (the prompt hint still asks the model to stay under
|
|
110
|
+
* `maxLength`) rather than writing a dead key no provider reads.
|
|
111
|
+
*
|
|
112
|
+
* Caveat (intentional): "caller wins" keys off ANY recognised spelling in
|
|
113
|
+
* `KNOWN_MAX_TOKENS_KEYS`, but only the adapter's native key is read on the
|
|
114
|
+
* wire. So a caller who sets a NON-native spelling for this provider — e.g.
|
|
115
|
+
* `maxTokens`, or Anthropic's `max_tokens` against an OpenAI adapter — suppresses
|
|
116
|
+
* the summarize default WITHOUT getting their own value applied either: neither
|
|
117
|
+
* cap reaches the wire. This favours never clobbering a migration leftover over
|
|
118
|
+
* guaranteeing a cap; the prompt-level hint still asks the model to stay under
|
|
119
|
+
* `maxLength`. Rename the key to the provider-native spelling to forward it.
|
|
120
|
+
*/
|
|
121
|
+
function applyMaxLength(
|
|
122
|
+
adapterName: string,
|
|
123
|
+
maxLength: number,
|
|
124
|
+
modelOptions: Record<string, unknown>,
|
|
125
|
+
): Record<string, unknown> {
|
|
126
|
+
const merged: Record<string, unknown> = { ...modelOptions }
|
|
127
|
+
|
|
128
|
+
if (adapterName === 'ollama') {
|
|
129
|
+
// Honor a caller-set limit in either shape: a recognised flat key (e.g.
|
|
130
|
+
// left over from a migration) or the nested `options.num_predict`.
|
|
131
|
+
const callerSetFlatLimit = KNOWN_MAX_TOKENS_KEYS.some(
|
|
132
|
+
(k) => typeof merged[k] === 'number',
|
|
133
|
+
)
|
|
134
|
+
const existing =
|
|
135
|
+
merged.options && typeof merged.options === 'object'
|
|
136
|
+
? (merged.options as Record<string, unknown>)
|
|
137
|
+
: undefined
|
|
138
|
+
if (
|
|
139
|
+
callerSetFlatLimit ||
|
|
140
|
+
(existing && typeof existing.num_predict === 'number')
|
|
141
|
+
) {
|
|
142
|
+
return merged
|
|
143
|
+
}
|
|
144
|
+
merged.options = { num_predict: maxLength, ...existing }
|
|
145
|
+
return merged
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const key = MAX_TOKENS_KEY_BY_ADAPTER[adapterName]
|
|
149
|
+
if (key === undefined) return merged
|
|
150
|
+
|
|
151
|
+
const callerSetLimit = KNOWN_MAX_TOKENS_KEYS.some(
|
|
152
|
+
(k) => typeof merged[k] === 'number',
|
|
153
|
+
)
|
|
154
|
+
if (callerSetLimit) return merged
|
|
155
|
+
|
|
156
|
+
merged[key] = maxLength
|
|
157
|
+
return merged
|
|
158
|
+
}
|
|
159
|
+
|
|
26
160
|
/**
|
|
27
161
|
* Extract the per-model `modelOptions` type a text adapter accepts. Used by
|
|
28
162
|
* provider summarize factories so their `modelOptions` IntelliSense matches
|
|
@@ -195,13 +329,38 @@ export class ChatStreamSummarizeAdapter<
|
|
|
195
329
|
options: SummarizationOptions<TProviderOptions>,
|
|
196
330
|
systemPrompt: string,
|
|
197
331
|
): TextOptions<TProviderOptions> {
|
|
332
|
+
// Sampling knobs now live in provider-native `modelOptions`. Apply the
|
|
333
|
+
// low-temperature default where the wrapped provider actually reads it
|
|
334
|
+
// (nested under `options` for Ollama, flat otherwise) so callers can still
|
|
335
|
+
// override it. Resolving the placement from this summarize adapter's OWN
|
|
336
|
+
// `name` keeps the default off the wire correctly per provider — a flat
|
|
337
|
+
// `temperature` would be silently dropped by Ollama while still showing up
|
|
338
|
+
// in OTel.
|
|
339
|
+
let working: Record<string, unknown> = {
|
|
340
|
+
...(options.modelOptions as Record<string, unknown> | undefined),
|
|
341
|
+
}
|
|
342
|
+
working = applyDefaultTemperature(this.name, 0.3, working)
|
|
343
|
+
// `maxLength` must reach the wire under the provider-native token key (it
|
|
344
|
+
// differs per provider, and no adapter reads a generic `maxTokens`).
|
|
345
|
+
// Resolve it from this summarize adapter's `name` (the constructor arg,
|
|
346
|
+
// not the wrapped text adapter's name), never overriding a caller-supplied
|
|
347
|
+
// token limit.
|
|
348
|
+
if (options.maxLength !== undefined) {
|
|
349
|
+
if (!isKnownMaxTokensAdapter(this.name)) {
|
|
350
|
+
options.logger.warn(
|
|
351
|
+
`summarize: maxLength=${options.maxLength} could not be mapped to a provider token key for adapter name "${this.name}" — it was dropped from modelOptions (the prompt still asks the model to stay under it). Construct ChatStreamSummarizeAdapter with a recognised provider name to forward the cap.`,
|
|
352
|
+
{ provider: this.name },
|
|
353
|
+
)
|
|
354
|
+
}
|
|
355
|
+
working = applyMaxLength(this.name, options.maxLength, working)
|
|
356
|
+
}
|
|
357
|
+
const modelOptions = working as TProviderOptions
|
|
358
|
+
|
|
198
359
|
return {
|
|
199
360
|
model: options.model,
|
|
200
361
|
messages: [{ role: 'user', content: options.text }],
|
|
201
362
|
systemPrompts: [systemPrompt],
|
|
202
|
-
|
|
203
|
-
temperature: 0.3,
|
|
204
|
-
modelOptions: options.modelOptions,
|
|
363
|
+
modelOptions,
|
|
205
364
|
logger: options.logger,
|
|
206
365
|
}
|
|
207
366
|
}
|
|
@@ -104,4 +104,22 @@ export class InternalLogger {
|
|
|
104
104
|
request(message: string, meta?: Record<string, unknown>): void {
|
|
105
105
|
this.emit('debug', 'request', message, meta)
|
|
106
106
|
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Log a non-fatal misconfiguration or recoverable anomaly. Gated by the
|
|
110
|
+
* `errors` category — on by default (and when `debug` is unspecified), so
|
|
111
|
+
* silent-drop conditions surface, but still silenced by `debug: false`,
|
|
112
|
+
* which honors the "disable everything including errors" contract. Routes to
|
|
113
|
+
* the underlying logger's `warn` level.
|
|
114
|
+
*/
|
|
115
|
+
warn(message: string, meta?: Record<string, unknown>): void {
|
|
116
|
+
if (!this.categories.errors) return
|
|
117
|
+
const prefixed = `⚠️ [tanstack-ai:warn] ⚠️ ${message}`
|
|
118
|
+
try {
|
|
119
|
+
this.logger.warn(prefixed, meta)
|
|
120
|
+
} catch {
|
|
121
|
+
// User-supplied logger threw; swallow so a broken logger never masks the
|
|
122
|
+
// condition we were trying to surface.
|
|
123
|
+
}
|
|
124
|
+
}
|
|
107
125
|
}
|
package/src/middlewares/otel.ts
CHANGED
|
@@ -4,6 +4,10 @@ import {
|
|
|
4
4
|
context as otelContext,
|
|
5
5
|
trace as otelTrace,
|
|
6
6
|
} from '@opentelemetry/api'
|
|
7
|
+
import {
|
|
8
|
+
MAX_TOKENS_KEYS,
|
|
9
|
+
NESTED_MAX_TOKENS_KEY,
|
|
10
|
+
} from '../utilities/sampling-keys'
|
|
7
11
|
import type {
|
|
8
12
|
AttributeValue,
|
|
9
13
|
Exception,
|
|
@@ -162,6 +166,19 @@ function messageEventName(role: string): string {
|
|
|
162
166
|
}
|
|
163
167
|
}
|
|
164
168
|
|
|
169
|
+
/**
|
|
170
|
+
* Return the first candidate that is a finite `number`, or `undefined`. Used to
|
|
171
|
+
* pick a sampling attribute from among the several provider-native spellings.
|
|
172
|
+
*/
|
|
173
|
+
function firstNumber(...candidates: Array<unknown>): number | undefined {
|
|
174
|
+
for (const candidate of candidates) {
|
|
175
|
+
if (typeof candidate === 'number' && Number.isFinite(candidate)) {
|
|
176
|
+
return candidate
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
return undefined
|
|
180
|
+
}
|
|
181
|
+
|
|
165
182
|
function errorMessage(err: unknown): string | undefined {
|
|
166
183
|
if (err instanceof Error) return err.message
|
|
167
184
|
if (typeof err === 'string') return err
|
|
@@ -333,12 +350,37 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
|
|
|
333
350
|
'gen_ai.request.model': ctx.model,
|
|
334
351
|
'tanstack.ai.iteration': ctx.iteration,
|
|
335
352
|
}
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
353
|
+
// Sampling options now live in provider-native `modelOptions`, and
|
|
354
|
+
// providers spell them differently (e.g. `max_output_tokens`,
|
|
355
|
+
// `max_completion_tokens`, `maxOutputTokens`, `num_predict`). Read the
|
|
356
|
+
// first numeric value among the known spellings — including Ollama's
|
|
357
|
+
// nested `options` — so gen_ai attributes populate across providers.
|
|
358
|
+
const sampling = config.modelOptions ?? {}
|
|
359
|
+
const nestedOptions =
|
|
360
|
+
sampling['options'] && typeof sampling['options'] === 'object'
|
|
361
|
+
? (sampling['options'] as Record<string, unknown>)
|
|
362
|
+
: undefined
|
|
363
|
+
const samplingTemperature = firstNumber(
|
|
364
|
+
sampling['temperature'],
|
|
365
|
+
nestedOptions?.['temperature'],
|
|
366
|
+
)
|
|
367
|
+
const samplingTopP = firstNumber(
|
|
368
|
+
sampling['top_p'],
|
|
369
|
+
sampling['topP'],
|
|
370
|
+
nestedOptions?.['top_p'],
|
|
371
|
+
)
|
|
372
|
+
// Spellings come from the shared `MAX_TOKENS_KEYS` table so this stays
|
|
373
|
+
// in lockstep with the summarize wrapper's caller-limit detection.
|
|
374
|
+
const samplingMaxTokens = firstNumber(
|
|
375
|
+
...MAX_TOKENS_KEYS.map((k) => sampling[k]),
|
|
376
|
+
nestedOptions?.[NESTED_MAX_TOKENS_KEY],
|
|
377
|
+
)
|
|
378
|
+
if (samplingTemperature !== undefined)
|
|
379
|
+
baseAttrs['gen_ai.request.temperature'] = samplingTemperature
|
|
380
|
+
if (samplingTopP !== undefined)
|
|
381
|
+
baseAttrs['gen_ai.request.top_p'] = samplingTopP
|
|
382
|
+
if (samplingMaxTokens !== undefined)
|
|
383
|
+
baseAttrs['gen_ai.request.max_tokens'] = samplingMaxTokens
|
|
342
384
|
|
|
343
385
|
const baseOptions: SpanOptions = {
|
|
344
386
|
kind: SpanKind.CLIENT,
|