@tanstack/ai 0.26.1 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/dist/esm/activities/chat/index.d.ts +7 -6
  2. package/dist/esm/activities/chat/index.js +78 -27
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/mcp/manager.d.ts +25 -0
  5. package/dist/esm/activities/chat/mcp/manager.js +71 -0
  6. package/dist/esm/activities/chat/mcp/manager.js.map +1 -0
  7. package/dist/esm/activities/chat/mcp/types.d.ts +56 -0
  8. package/dist/esm/activities/chat/middleware/types.d.ts +1 -4
  9. package/dist/esm/activities/chat/stream/message-updaters.js +20 -8
  10. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -1
  11. package/dist/esm/activities/chat/tools/tool-calls.d.ts +1 -1
  12. package/dist/esm/activities/chat/tools/tool-calls.js +2 -1
  13. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  14. package/dist/esm/activities/summarize/chat-stream-summarize.js +62 -3
  15. package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
  16. package/dist/esm/extend-adapter.d.ts +22 -6
  17. package/dist/esm/extend-adapter.js.map +1 -1
  18. package/dist/esm/index.d.ts +2 -0
  19. package/dist/esm/index.js +2 -0
  20. package/dist/esm/index.js.map +1 -1
  21. package/dist/esm/logger/internal-logger.d.ts +8 -0
  22. package/dist/esm/logger/internal-logger.js +15 -0
  23. package/dist/esm/logger/internal-logger.js.map +1 -1
  24. package/dist/esm/middlewares/otel.js +30 -6
  25. package/dist/esm/middlewares/otel.js.map +1 -1
  26. package/dist/esm/types.d.ts +6 -35
  27. package/dist/esm/utilities/sampling-keys.d.ts +20 -0
  28. package/dist/esm/utilities/sampling-keys.js +20 -0
  29. package/dist/esm/utilities/sampling-keys.js.map +1 -0
  30. package/package.json +2 -2
  31. package/skills/ai-core/adapter-configuration/SKILL.md +67 -6
  32. package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +6 -3
  33. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +3 -0
  34. package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +10 -1
  35. package/skills/ai-core/adapter-configuration/references/openai-adapter.md +4 -0
  36. package/skills/ai-core/chat-experience/SKILL.md +95 -7
  37. package/skills/ai-core/middleware/SKILL.md +11 -0
  38. package/skills/ai-core/tool-calling/SKILL.md +287 -0
  39. package/src/activities/chat/index.ts +97 -35
  40. package/src/activities/chat/mcp/manager.ts +85 -0
  41. package/src/activities/chat/mcp/types.ts +66 -0
  42. package/src/activities/chat/middleware/types.ts +1 -4
  43. package/src/activities/chat/stream/message-updaters.ts +22 -9
  44. package/src/activities/chat/tools/tool-calls.ts +2 -0
  45. package/src/activities/summarize/chat-stream-summarize.ts +162 -3
  46. package/src/extend-adapter.ts +42 -24
  47. package/src/index.ts +10 -0
  48. package/src/logger/internal-logger.ts +18 -0
  49. package/src/middlewares/otel.ts +48 -6
  50. package/src/types.ts +6 -35
  51. package/src/utilities/sampling-keys.ts +28 -0
@@ -348,6 +348,12 @@ type RuntimeContextField<TContext> = IsUnknown<TContext> extends true ? {
348
348
  export type ToolExecutionContext<TContext = unknown> = RuntimeContextField<TContext> & {
349
349
  /** The ID of the tool call being executed */
350
350
  toolCallId?: string;
351
+ /**
352
+ * Abort signal for the current chat run. Aborts when the run's
353
+ * `abortController` fires (or middleware aborts). Long-running tools —
354
+ * e.g. MCP `callTool` — should forward this to cancel in-flight work.
355
+ */
356
+ abortSignal?: AbortSignal;
351
357
  /**
352
358
  * Emit a custom event during tool execution.
353
359
  * Events are streamed to the client in real-time as AG-UI CUSTOM events.
@@ -629,41 +635,6 @@ export interface TextOptions<TProviderOptionsSuperset extends Record<string, any
629
635
  */
630
636
  systemPrompts?: Array<SystemPrompt>;
631
637
  agentLoopStrategy?: AgentLoopStrategy;
632
- /**
633
- * Controls the randomness of the output.
634
- * Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
635
- * Range: [0.0, 2.0]
636
- *
637
- * Note: Generally recommended to use either temperature or topP, but not both.
638
- *
639
- * Provider usage:
640
- * - OpenAI: `temperature` (number) - in text.top_p field
641
- * - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
642
- * - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
643
- */
644
- temperature?: number;
645
- /**
646
- * Nucleus sampling parameter. An alternative to temperature sampling.
647
- * The model considers the results of tokens with topP probability mass.
648
- * For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
649
- *
650
- * Note: Generally recommended to use either temperature or topP, but not both.
651
- *
652
- * Provider usage:
653
- * - OpenAI: `text.top_p` (number)
654
- * - Anthropic: `top_p` (number | null)
655
- * - Gemini: `generationConfig.topP` (number)
656
- */
657
- topP?: number;
658
- /**
659
- * The maximum number of tokens to generate in the response.
660
- *
661
- * Provider usage:
662
- * - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
663
- * - Anthropic: `max_tokens` (number, required) - range x >= 1
664
- * - Gemini: `generationConfig.maxOutputTokens` (number)
665
- */
666
- maxTokens?: number;
667
638
  /**
668
639
  * Additional metadata to attach to the request.
669
640
  * Can be used for tracking, debugging, or passing custom information.
@@ -0,0 +1,20 @@
1
+ /**
2
+ * Single source of truth for the provider-native key spellings that cap output
3
+ * tokens. Sampling options live in opaque, provider-native `modelOptions`, and
4
+ * every provider spells the token cap differently. Two call sites must agree on
5
+ * this set or they silently drift:
6
+ *
7
+ * - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied
8
+ * token limit so the summarize default never overrides it.
9
+ * - `middlewares/otel.ts` — picks the first numeric spelling to populate the
10
+ * `gen_ai.request.max_tokens` attribute across providers.
11
+ *
12
+ * Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →
13
+ * native-key map) in the summarize wrapper.
14
+ */
15
+ export declare const MAX_TOKENS_KEYS: readonly ["max_output_tokens", "max_tokens", "max_completion_tokens", "maxOutputTokens", "maxCompletionTokens", "maxTokens"];
16
+ /**
17
+ * Ollama nests sampling under `options`; its token cap is `options.num_predict`
18
+ * rather than a flat key.
19
+ */
20
+ export declare const NESTED_MAX_TOKENS_KEY: "num_predict";
@@ -0,0 +1,20 @@
1
+ const MAX_TOKENS_KEYS = [
2
+ "max_output_tokens",
3
+ // OpenAI (Responses)
4
+ "max_tokens",
5
+ // Anthropic / Grok
6
+ "max_completion_tokens",
7
+ // Groq
8
+ "maxOutputTokens",
9
+ // Gemini
10
+ "maxCompletionTokens",
11
+ // OpenRouter
12
+ "maxTokens"
13
+ // generic / migration leftover (no adapter reads it)
14
+ ];
15
+ const NESTED_MAX_TOKENS_KEY = "num_predict";
16
+ export {
17
+ MAX_TOKENS_KEYS,
18
+ NESTED_MAX_TOKENS_KEY
19
+ };
20
+ //# sourceMappingURL=sampling-keys.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sampling-keys.js","sources":["../../../src/utilities/sampling-keys.ts"],"sourcesContent":["/**\n * Single source of truth for the provider-native key spellings that cap output\n * tokens. Sampling options live in opaque, provider-native `modelOptions`, and\n * every provider spells the token cap differently. Two call sites must agree on\n * this set or they silently drift:\n *\n * - `activities/summarize/chat-stream-summarize.ts` — detects a caller-supplied\n * token limit so the summarize default never overrides it.\n * - `middlewares/otel.ts` — picks the first numeric spelling to populate the\n * `gen_ai.request.max_tokens` attribute across providers.\n *\n * Keep this list in lockstep with `MAX_TOKENS_KEY_BY_ADAPTER` (the adapter →\n * native-key map) in the summarize wrapper.\n */\nexport const MAX_TOKENS_KEYS = [\n 'max_output_tokens', // OpenAI (Responses)\n 'max_tokens', // Anthropic / Grok\n 'max_completion_tokens', // Groq\n 'maxOutputTokens', // Gemini\n 'maxCompletionTokens', // OpenRouter\n 'maxTokens', // generic / migration leftover (no adapter reads it)\n] as const\n\n/**\n * Ollama nests sampling under `options`; its token cap is `options.num_predict`\n * rather than a flat key.\n */\nexport const NESTED_MAX_TOKENS_KEY = 'num_predict' as const\n"],"names":[],"mappings":"AAcO,MAAM,kBAAkB;AAAA,EAC7B;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AACF;AAMO,MAAM,wBAAwB;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.26.1",
3
+ "version": "0.28.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -68,7 +68,7 @@
68
68
  "@ag-ui/core": "^0.0.52",
69
69
  "@standard-schema/spec": "^1.1.0",
70
70
  "partial-json": "^0.1.7",
71
- "@tanstack/ai-event-client": "0.5.2"
71
+ "@tanstack/ai-event-client": "0.5.4"
72
72
  },
73
73
  "peerDependencies": {
74
74
  "@opentelemetry/api": ">=1.9.0"
@@ -44,8 +44,10 @@ import { openaiText } from '@tanstack/ai-openai'
44
44
  const stream = chat({
45
45
  adapter: openaiText('gpt-5.2'),
46
46
  messages,
47
- temperature: 0.7,
48
- maxTokens: 1000,
47
+ modelOptions: {
48
+ temperature: 0.7,
49
+ max_output_tokens: 1000,
50
+ },
49
51
  })
50
52
 
51
53
  return toServerSentEventsResponse(stream)
@@ -55,6 +57,11 @@ The adapter factory function takes the model name as a string literal and an
55
57
  optional config object (API key, base URL, etc.). The model name is passed
56
58
  into the factory, not into `chat()`.
57
59
 
60
+ Sampling options (`temperature`, token limits, `top_p`/`topP`, etc.) live
61
+ inside `modelOptions` using each provider's native key — they are **not**
62
+ top-level options on `chat()`. See the per-provider table in
63
+ [Configuring Sampling](#5-configuring-sampling) below.
64
+
58
65
  ## Core Patterns
59
66
 
60
67
  ### 1. Adapter Selection
@@ -158,11 +165,11 @@ const openaiStream = chat({
158
165
  const anthropicStream = chat({
159
166
  adapter: anthropicText('claude-sonnet-4-6'),
160
167
  messages,
161
- maxTokens: 16000,
162
168
  modelOptions: {
169
+ max_tokens: 16000,
163
170
  thinking: {
164
171
  type: 'enabled',
165
- budget_tokens: 8000, // must be >= 1024 and < maxTokens
172
+ budget_tokens: 8000, // must be >= 1024 and < max_tokens
166
173
  },
167
174
  },
168
175
  })
@@ -171,8 +178,8 @@ const anthropicStream = chat({
171
178
  const adaptiveStream = chat({
172
179
  adapter: anthropicText('claude-sonnet-4-6'),
173
180
  messages,
174
- maxTokens: 16000,
175
181
  modelOptions: {
182
+ max_tokens: 16000,
176
183
  thinking: {
177
184
  type: 'adaptive',
178
185
  },
@@ -224,7 +231,61 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
224
231
  At runtime, `extendAdapter` simply passes through to the original factory.
225
232
  The `_customModels` parameter is only used for type inference.
226
233
 
227
- ### 5. Capability Flag: `supportsCombinedToolsAndSchema`
234
+ ### 5. Configuring Sampling
235
+
236
+ Sampling controls (`temperature`, token limits, nucleus sampling) are passed
237
+ inside `modelOptions` using each provider's **native** key. They are not
238
+ top-level fields on `chat()`/`ai()`/`generate()`.
239
+
240
+ ```typescript
241
+ // OpenAI — native keys
242
+ chat({
243
+ adapter: openaiText('gpt-5.2'),
244
+ messages,
245
+ modelOptions: { temperature: 0.7, top_p: 0.9, max_output_tokens: 1000 },
246
+ })
247
+
248
+ // Anthropic
249
+ chat({
250
+ adapter: anthropicText('claude-sonnet-4-6'),
251
+ messages,
252
+ modelOptions: { temperature: 0.7, top_p: 0.9, max_tokens: 1000 },
253
+ })
254
+
255
+ // Gemini — camelCase
256
+ chat({
257
+ adapter: geminiText('gemini-2.5-pro'),
258
+ messages,
259
+ modelOptions: { temperature: 0.7, topP: 0.9, maxOutputTokens: 1000 },
260
+ })
261
+
262
+ // Ollama — NESTED under modelOptions.options
263
+ chat({
264
+ adapter: ollamaText('llama3.3'),
265
+ messages,
266
+ modelOptions: {
267
+ options: { temperature: 0.7, top_p: 0.9, num_predict: 1000 },
268
+ },
269
+ })
270
+ ```
271
+
272
+ Per-provider sampling keys (all live inside `modelOptions`):
273
+
274
+ | Provider | Temperature | Nucleus | Max output tokens |
275
+ | ----------------- | ------------- | ------- | ----------------------------------- |
276
+ | OpenAI | `temperature` | `top_p` | `max_output_tokens` |
277
+ | Anthropic | `temperature` | `top_p` | `max_tokens` |
278
+ | Gemini | `temperature` | `topP` | `maxOutputTokens` |
279
+ | Grok (xAI) | `temperature` | `top_p` | `max_tokens` |
280
+ | Groq | `temperature` | `top_p` | `max_completion_tokens` |
281
+ | OpenRouter (chat) | `temperature` | `topP` | `maxCompletionTokens` |
282
+ | Ollama | `temperature` | `top_p` | `num_predict` (nested in `options`) |
283
+
284
+ `temperature` is the one key every provider names identically; token limits and
285
+ some sampling options use provider-native names. Ollama nests all sampling under
286
+ `modelOptions.options`.
287
+
288
+ ### 6. Capability Flag: `supportsCombinedToolsAndSchema`
228
289
 
229
290
  Adapters can declare an optional capability method:
230
291
 
@@ -39,12 +39,15 @@ Note: Model IDs use the format `claude-opus-4-6`, `claude-sonnet-4-6`, etc.
39
39
  chat({
40
40
  adapter: anthropicText('claude-sonnet-4-6'),
41
41
  messages,
42
- maxTokens: 16000,
43
42
  modelOptions: {
43
+ // Sampling
44
+ temperature: 0.7,
45
+ top_p: 0.9, // cannot be combined with temperature
46
+ max_tokens: 16000,
44
47
  // Extended thinking (budget-based)
45
48
  thinking: {
46
49
  type: 'enabled',
47
- budget_tokens: 8000, // must be >= 1024 and < maxTokens
50
+ budget_tokens: 8000, // must be >= 1024 and < max_tokens
48
51
  },
49
52
  // Adaptive thinking (claude-sonnet-4-6, claude-opus-4-6+)
50
53
  thinking: {
@@ -89,7 +92,7 @@ ANTHROPIC_API_KEY
89
92
 
90
93
  ## Gotchas
91
94
 
92
- - `thinking.budget_tokens` must be >= 1024 AND less than `maxTokens`.
95
+ - `thinking.budget_tokens` must be >= 1024 AND less than `modelOptions.max_tokens`.
93
96
  Failing either check throws a validation error.
94
97
  - Cannot set both `top_p` and `temperature` at the same time (throws error).
95
98
  - `claude-3-5-haiku` and `claude-3-haiku` do NOT support extended thinking.
@@ -72,6 +72,9 @@ chat({
72
72
  // Response modalities
73
73
  responseModalities: ['TEXT'],
74
74
  // Sampling
75
+ temperature: 0.7,
76
+ topP: 0.9,
77
+ maxOutputTokens: 1000,
75
78
  topK: 40,
76
79
  seed: 42,
77
80
  presencePenalty: 0.5,
@@ -40,6 +40,9 @@ Models must be pulled first: `ollama pull llama3.3`
40
40
 
41
41
  Ollama models use a generic options type. Provider options vary by the
42
42
  underlying model. The adapter passes options through to the Ollama API.
43
+ Sampling options are **nested** under `modelOptions.options` (this matches
44
+ Ollama's own request shape) — `temperature`, `top_p`, and `num_predict`
45
+ (max output tokens) all live there.
43
46
 
44
47
  ```typescript
45
48
  import { chat } from '@tanstack/ai'
@@ -48,7 +51,13 @@ import { ollamaText } from '@tanstack/ai-ollama'
48
51
  const stream = chat({
49
52
  adapter: ollamaText('llama3.3'),
50
53
  messages,
51
- temperature: 0.7,
54
+ modelOptions: {
55
+ options: {
56
+ temperature: 0.7,
57
+ top_p: 0.9,
58
+ num_predict: 1000, // max output tokens
59
+ },
60
+ },
52
61
  // Ollama-specific options are limited compared to cloud providers
53
62
  })
54
63
  ```
@@ -44,6 +44,10 @@ chat({
44
44
  adapter: openaiText('gpt-5.4'),
45
45
  messages,
46
46
  modelOptions: {
47
+ // Sampling
48
+ temperature: 0.7,
49
+ top_p: 0.9,
50
+ max_output_tokens: 1000,
47
51
  // Reasoning (effort levels: none, minimal, low, medium, high)
48
52
  reasoning: {
49
53
  effort: 'high',
@@ -137,8 +137,10 @@ import { anthropicText } from '@tanstack/ai-anthropic'
137
137
  const stream = chat({
138
138
  adapter: anthropicText('claude-sonnet-4-5'),
139
139
  messages,
140
- temperature: 0.7,
141
- maxTokens: 2000,
140
+ modelOptions: {
141
+ temperature: 0.7,
142
+ max_tokens: 2000, // Anthropic-native key
143
+ },
142
144
  systemPrompts: ['You are a helpful assistant.'],
143
145
  abortController,
144
146
  })
@@ -308,6 +310,77 @@ const { messages, sendMessage } = useChat({
308
310
  The only difference is swapping `toServerSentEventsResponse` / `fetchServerSentEvents`
309
311
  for `toHttpResponse` / `fetchHttpStream`. Everything else stays identical.
310
312
 
313
+ ### 5. MCP Tool Discovery via `chat({ mcp })`
314
+
315
+ Pass `mcp` to let `chat()` own discovery **and** lifecycle for one or more MCP
316
+ clients. Useful when you want minimal boilerplate and don't need to reuse the
317
+ clients across calls.
318
+
319
+ ```typescript
320
+ // Prop shape:
321
+ // chat({
322
+ // ...,
323
+ // mcp: {
324
+ // clients: Array<MCPClient | MCPClients>,
325
+ // connection?: 'close' | 'keep-alive', // default: 'close'
326
+ // lazyTools?: boolean,
327
+ // onDiscoveryError?: (error: unknown, source) => void,
328
+ // }
329
+ // })
330
+ ```
331
+
332
+ - **`clients`** — one or more `MCPClient` / `MCPClients` instances.
333
+ - **`connection`** — `'close'` (default) closes each client when the run ends
334
+ (after the agent loop completes and the stream is drained); with
335
+ `'keep-alive'`, `chat()` never closes the clients — the caller owns their
336
+ lifecycle (keep connections warm across requests).
337
+ - **`lazyTools`** — forwarded to `tools({ lazy: true })` so tool schemas are
338
+ sent to the LLM on demand.
339
+ - **`onDiscoveryError`** — throw (or re-throw) to fail the entire call fast;
340
+ return normally to skip that source and continue. Omit to rethrow (fail-fast).
341
+
342
+ **When to use `mcp` vs. the tools spread:**
343
+
344
+ | Approach | Use when |
345
+ | ------------------------------------------------------- | ----------------------------------------------------------------------------------------------- |
346
+ | `chat({ mcp: { clients: [...] } })` | You want discovery + lifecycle managed for you, and don't need fully-typed input/output schemas |
347
+ | `tools: [...await client.tools([toolDefinition(...)])]` | You want fully-typed MCP tools with Zod input/output validation |
348
+
349
+ **Server-side example:**
350
+
351
+ ```typescript
352
+ import { createFileRoute } from '@tanstack/react-router'
353
+ import { chat, toServerSentEventsResponse } from '@tanstack/ai'
354
+ import { openaiText } from '@tanstack/ai-openai'
355
+ import { createMCPClient } from '@tanstack/ai-mcp'
356
+
357
+ export const Route = createFileRoute('/api/chat')({
358
+ server: {
359
+ handlers: {
360
+ POST: async ({ request }) => {
361
+ const { messages } = await request.json()
362
+
363
+ const mcpClient = await createMCPClient({
364
+ transport: { type: 'http', url: 'https://mcp.example.com/mcp' },
365
+ })
366
+
367
+ const stream = chat({
368
+ adapter: openaiText('gpt-5.5'),
369
+ messages,
370
+ mcp: {
371
+ clients: [mcpClient],
372
+ connection: 'keep-alive', // chat() won't close it — reuse across requests
373
+ },
374
+ })
375
+
376
+ return toServerSentEventsResponse(stream)
377
+ // connection: 'keep-alive' — chat() never closes mcpClient; it stays open for reuse across runs.
378
+ },
379
+ },
380
+ },
381
+ })
382
+ ```
383
+
311
384
  ## Common Mistakes
312
385
 
313
386
  ### a. CRITICAL: Using Vercel AI SDK patterns (streamText, generateText)
@@ -377,17 +450,32 @@ chat({ adapter: openaiText('gpt-5.2'), messages })
377
450
 
378
451
  The model is passed to the adapter factory, not to `chat()`.
379
452
 
380
- ### f. HIGH: Nesting temperature/maxTokens in options object
453
+ ### f. HIGH: Passing sampling options at the root of chat()
454
+
455
+ Sampling options (`temperature`, token limits, `top_p`/`topP`) are **not**
456
+ top-level fields on `chat()`. They live inside `modelOptions` using the
457
+ provider's native key.
381
458
 
382
459
  ```typescript
383
- // WRONG
460
+ // WRONG — temperature/maxTokens are not root options
461
+ chat({ adapter, messages, temperature: 0.7, maxTokens: 1000 })
462
+
463
+ // WRONG — there is no `options` field either
384
464
  chat({ adapter, messages, options: { temperature: 0.7, maxTokens: 1000 } })
385
465
 
386
- // CORRECT
387
- chat({ adapter, messages, temperature: 0.7, maxTokens: 1000 })
466
+ // CORRECT — inside modelOptions, provider-native keys (OpenAI shown)
467
+ chat({
468
+ adapter,
469
+ messages,
470
+ modelOptions: { temperature: 0.7, max_output_tokens: 1000 },
471
+ })
388
472
  ```
389
473
 
390
- All parameters are top-level on the `chat()` options object.
474
+ `temperature` is universal across providers; token limits use provider-native
475
+ keys (`max_output_tokens` for OpenAI, `max_tokens` for Anthropic/Grok,
476
+ `maxOutputTokens` for Gemini, `max_completion_tokens` for Groq,
477
+ `maxCompletionTokens` for OpenRouter, and `num_predict` nested under
478
+ `modelOptions.options` for Ollama). See ai-core/adapter-configuration/SKILL.md.
391
479
 
392
480
  ### g. HIGH: Using providerOptions instead of modelOptions
393
481
 
@@ -68,6 +68,13 @@ Every hook receives a `ChatMiddlewareContext` as its first argument, which provi
68
68
  Terminal hooks (`onFinish`, `onAbort`, `onError`) are **mutually exclusive** -- exactly
69
69
  one fires per `chat()` invocation.
70
70
 
71
+ > **Sampling in `onConfig`:** `temperature`, `topP`, and `maxTokens` are **not**
72
+ > first-class fields on `ChatMiddlewareConfig`. To adjust sampling from
73
+ > middleware, return a partial that mutates `config.modelOptions` using the
74
+ > provider's native key (e.g. OpenAI `temperature` / `max_output_tokens`,
75
+ > Anthropic `max_tokens`, Ollama nested `options.num_predict`). Returning a
76
+ > top-level `temperature`/`maxTokens` has no effect.
77
+
71
78
  ### Phase values
72
79
 
73
80
  `ctx.phase` is one of:
@@ -304,6 +311,10 @@ const configTransform: ChatMiddleware = {
304
311
  if (ctx.phase === 'init') {
305
312
  return {
306
313
  systemPrompts: [...config.systemPrompts, 'Always respond in JSON.'],
314
+ // Sampling options are NOT first-class config fields — mutate them
315
+ // through `config.modelOptions` using the provider's native key.
316
+ // (e.g. OpenAI `temperature` / `max_output_tokens`.)
317
+ modelOptions: { ...config.modelOptions, temperature: 0.2 },
307
318
  }
308
319
  }
309
320
  },