@owlmeans/llm 0.1.18-rc.2 → 0.1.18-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +2 -2
  2. package/agent-meta/manifest.json +2 -2
  3. package/agent-meta/skills/llm/SKILL.md +239 -116
  4. package/agent-meta/skills/llm-prompt-caching/SKILL.md +37 -3
  5. package/build/execution/service.d.ts.map +1 -1
  6. package/build/execution/service.js +46 -11
  7. package/build/execution/service.js.map +1 -1
  8. package/build/execution/types.d.ts +72 -12
  9. package/build/execution/types.d.ts.map +1 -1
  10. package/build/execution/utils.d.ts.map +1 -1
  11. package/build/execution/utils.js +15 -9
  12. package/build/execution/utils.js.map +1 -1
  13. package/build/helpers/retry.d.ts +14 -0
  14. package/build/helpers/retry.d.ts.map +1 -1
  15. package/build/helpers/retry.js +14 -0
  16. package/build/helpers/retry.js.map +1 -1
  17. package/build/helpers/spectate.d.ts.map +1 -1
  18. package/build/helpers/spectate.js +12 -5
  19. package/build/helpers/spectate.js.map +1 -1
  20. package/build/index.d.ts +2 -2
  21. package/build/index.d.ts.map +1 -1
  22. package/build/index.js +2 -2
  23. package/build/index.js.map +1 -1
  24. package/build/model.d.ts +1 -1
  25. package/build/model.d.ts.map +1 -1
  26. package/build/model.js +60 -28
  27. package/build/model.js.map +1 -1
  28. package/build/plugins/anthropic.d.ts +40 -0
  29. package/build/plugins/anthropic.d.ts.map +1 -1
  30. package/build/plugins/anthropic.js +79 -6
  31. package/build/plugins/anthropic.js.map +1 -1
  32. package/build/plugins/openai.d.ts +12 -0
  33. package/build/plugins/openai.d.ts.map +1 -1
  34. package/build/plugins/openai.js +29 -3
  35. package/build/plugins/openai.js.map +1 -1
  36. package/build/plugins/types.d.ts +7 -0
  37. package/build/plugins/types.d.ts.map +1 -1
  38. package/build/prompt/service.d.ts.map +1 -1
  39. package/build/prompt/service.js +11 -0
  40. package/build/prompt/service.js.map +1 -1
  41. package/build/prompt/types.d.ts +20 -0
  42. package/build/prompt/types.d.ts.map +1 -1
  43. package/build/service.d.ts.map +1 -1
  44. package/build/service.js +42 -7
  45. package/build/service.js.map +1 -1
  46. package/build/types.d.ts +71 -1
  47. package/build/types.d.ts.map +1 -1
  48. package/build/utils/config.d.ts +11 -0
  49. package/build/utils/config.d.ts.map +1 -1
  50. package/build/utils/config.js +21 -1
  51. package/build/utils/config.js.map +1 -1
  52. package/build/utils/null-report.d.ts.map +1 -1
  53. package/build/utils/null-report.js +9 -1
  54. package/build/utils/null-report.js.map +1 -1
  55. package/package.json +13 -13
  56. package/src/execution/service.ts +46 -11
  57. package/src/execution/types.ts +77 -13
  58. package/src/execution/utils.ts +16 -9
  59. package/src/helpers/retry.ts +16 -0
  60. package/src/helpers/spectate.ts +14 -5
  61. package/src/index.ts +4 -2
  62. package/src/model.ts +77 -27
  63. package/src/plugins/anthropic.ts +89 -6
  64. package/src/plugins/openai.ts +33 -3
  65. package/src/plugins/types.ts +8 -0
  66. package/src/prompt/service.ts +12 -0
  67. package/src/prompt/types.ts +20 -0
  68. package/src/service.ts +53 -7
  69. package/src/types.ts +71 -1
  70. package/src/utils/config.ts +23 -1
  71. package/src/utils/null-report.ts +9 -1
  72. package/tests/context.ts +3 -0
  73. package/tests/execution.spec.ts +148 -13
  74. package/tests/helpers.spec.ts +4 -1
  75. package/tests/plugins.spec.ts +211 -0
  76. package/tests/prompt.spec.ts +43 -0
@@ -5,6 +5,15 @@ import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
5
5
  import type { LlmSpectator, ModelInputItem } from '../types.js'
6
6
  import { hasCacheActivity, readCacheUsage } from './cache.js'
7
7
 
8
+ /**
9
+ * The trace's `content` column is a string; LangChain content is a string-or-blocks union, and
10
+ * tool calls are arrays. Anything non-string must be stored as JSON — a plain string coercion
11
+ * downstream turns it into `[object Object]` and destroys the readable half of the trace
12
+ * (`raw` keeps the payload, but `content` is what trace readers query).
13
+ */
14
+ const asText = (content: unknown): string =>
15
+ typeof content === 'string' ? content : JSON.stringify(content) ?? ''
16
+
8
17
  /** Normalize one prompt message into the spectator's storage shape. */
9
18
  const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
10
19
  const entry: SpectatorEntryMessage = {
@@ -16,17 +25,17 @@ const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMes
16
25
 
17
26
  if (msg instanceof BaseMessage) {
18
27
  entry.type = msg.type
19
- entry.content = msg.content as unknown as string
28
+ entry.content = asText(msg.content)
20
29
  entry.name = msg.name
21
30
  entry.contentType = typeof msg.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json
22
31
  entry.usage = 'usage_metadata' in msg ? msg.usage_metadata as UsageMetadata : undefined
23
32
  entry.raw = JSON.stringify({ message: msg, content: msg.content, meta: msg.response_metadata })
24
33
  } else if (typeof msg === 'object' && 'content' in msg) {
25
34
  entry.type = msg.role as string ?? 'unknown'
26
- entry.content = msg.content as string ?? msg as unknown as string
35
+ entry.content = asText(msg.content ?? msg)
27
36
  entry.raw = JSON.stringify(msg)
28
37
  } else {
29
- entry.content = msg as unknown as string
38
+ entry.content = asText(msg)
30
39
  entry.raw = JSON.stringify(msg)
31
40
  }
32
41
 
@@ -58,10 +67,10 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
58
67
  }
59
68
 
60
69
  if (message.tool_calls != null && message.tool_calls.length > 0) {
61
- completion.content = message.tool_calls as unknown as string
70
+ completion.content = asText(message.tool_calls)
62
71
  completion.contentType = SpectatorContentType.ToolCall
63
72
  } else {
64
- completion.content = message.content as unknown as string
73
+ completion.content = asText(message.content)
65
74
  }
66
75
 
67
76
  // Silent unless the provider reported cache activity, so it costs nothing when
package/src/index.ts CHANGED
@@ -9,6 +9,8 @@ export * from './execution/index.js'
9
9
  export * from './prompt/index.js'
10
10
  export type * from './plugins/types.js'
11
11
  export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
12
- export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
12
+ export { anthropicPlugin, ANTHROPIC_FAMILY, NO_SAMPLING_PREFIXES, rejectsSampling } from './plugins/anthropic.js'
13
13
  export { compatiblePlugin } from './plugins/compatible.js'
14
- export { openAiPlugin, openAiFamily, OPENAI_FAMILY } from './plugins/openai.js'
14
+ export {
15
+ openAiPlugin, openAiFamily, OPENAI_FAMILY, RESPONSES_API_PREFIXES, usesResponsesApi,
16
+ } from './plugins/openai.js'
package/src/model.ts CHANGED
@@ -6,7 +6,7 @@ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
6
6
  import { StructuredMode } from '@owlmeans/llm-common'
7
7
  import type { NullKind } from '@owlmeans/llm-common'
8
8
  import {
9
- DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
9
+ DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
10
10
  } from './consts.js'
11
11
  import { LlmModelError } from './errors.js'
12
12
  import { pluginFor, pluginOf } from './plugins/index.js'
@@ -15,7 +15,7 @@ import { coerceToSchema, parseJsonContent } from './helpers/json.js'
15
15
  import { normalizeInput } from './helpers/messages.js'
16
16
  import { withRetry } from './helpers/retry.js'
17
17
  import { spectate } from './helpers/spectate.js'
18
- import { idleTimeout, readConfig } from './utils/config.js'
18
+ import { idleTimeout, readConfig, resolveOutputCap } from './utils/config.js'
19
19
  import { reportNull } from './utils/null-report.js'
20
20
  import type { NullReportParams } from './utils/null-report.js'
21
21
  import { applyNoThink, dropBlankContent, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
@@ -88,6 +88,7 @@ export const makeLlmModel = ({
88
88
  prompt,
89
89
  prompts,
90
90
  files,
91
+ utility,
91
92
  }: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
92
93
 
93
94
  const ajv = new Ajv({ strict: false })
@@ -98,6 +99,17 @@ export const makeLlmModel = ({
98
99
  const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
99
100
  const timeout = idleTimeout(config)
100
101
 
102
+ /**
103
+ * Where on the escalation ladder this call starts.
104
+ *
105
+ * The ladder has exactly `retries` rungs, so a seed past the last one buys nothing and
106
+ * would only inflate the attempt number handed to `refine`. Clamped, `refineModel` sees
107
+ * at most `2 * (retries - 1)` — the escalator's own doubling stays bounded by the
108
+ * output cap either way.
109
+ */
110
+ const ladderSeed = (escalation?: number): number =>
111
+ Math.max(0, Math.min(Math.floor(escalation ?? 0), retries - 1))
112
+
101
113
  /**
102
114
  * Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
103
115
  * dependency order.
@@ -133,7 +145,7 @@ export const makeLlmModel = ({
133
145
  callSkills: callSkills ?? prompt?.callSkills,
134
146
  },
135
147
  msgs,
136
- { model, provider: plugin, purpose, action, cacheMax, files },
148
+ { model, provider: plugin, purpose, action, cacheMax, files, utility },
137
149
  )
138
150
  if (composed.system != null) {
139
151
  msgs.unshift(composed.system)
@@ -147,7 +159,9 @@ export const makeLlmModel = ({
147
159
  }
148
160
 
149
161
  if (json) ensureJsonMention(msgs)
150
- applyNoThink(msgs, config.disableThinking)
162
+ // The soft switch is for models with no request-level control; a plugin that sends the
163
+ // real parameter must not also get the directive as prompt text.
164
+ applyNoThink(msgs, config.disableThinking === true && plugin?.suppressesThinking?.(config) !== true)
151
165
  // Cache markers replace string content with content blocks, so they must go last.
152
166
  const ttl = prompt?.cacheTtl
153
167
  const marked = plugin?.patchCache?.(msgs, {
@@ -216,9 +230,9 @@ export const makeLlmModel = ({
216
230
  const basePlugin = pluginOf(baseConfig.provider) ?? pluginFor(base)
217
231
  if (basePlugin == null) return base
218
232
 
219
- const maxOutputCap = typeof baseConfig.maxTokensCap === 'number' && baseConfig.maxTokensCap > 0
220
- ? baseConfig.maxTokensCap
221
- : DEFAULT_MAX_OUTPUT_CAP
233
+ // Read from the ACTIVE base — after the fallback swap that is the fallback's own
234
+ // config, so the escalator sizes the model it is actually talking to.
235
+ const maxOutputCap = resolveOutputCap(baseConfig)
222
236
  const refined = basePlugin.refine({ base, attempt, temperature, maxOutputCap })
223
237
 
224
238
  if (attempt > 0) {
@@ -297,11 +311,15 @@ export const makeLlmModel = ({
297
311
  const helper: LlmModel = {
298
312
  ask: async (
299
313
  input,
300
- { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmAskOptions
314
+ {
315
+ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
316
+ escalation, fatal,
317
+ }: LlmAskOptions
301
318
  ) => {
302
319
  const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
303
- return withRetry({ retries, outputErrors }, async i => {
304
- const refined = refineModel(i)
320
+ const seed = ladderSeed(escalation)
321
+ return withRetry({ retries, outputErrors, fatal }, async i => {
322
+ const refined = refineModel(seed + i)
305
323
  console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
306
324
  const startedAt = Date.now()
307
325
  let result: AIMessageChunk | null = null
@@ -315,7 +333,7 @@ export const makeLlmModel = ({
315
333
  }
316
334
 
317
335
  const message = new AIMessage(result)
318
- let output: string | null = typeof result.content === 'string'
336
+ let output: string = typeof result.content === 'string'
319
337
  ? result.content
320
338
  : Array.isArray(result.content)
321
339
  ? result.content
@@ -323,19 +341,39 @@ export const makeLlmModel = ({
323
341
  typeof c === 'object' && c !== null && 'type' in c && c.type === 'text'
324
342
  )
325
343
  .map(c => c.text)
326
- .join('') || null
327
- : null
344
+ .join('')
345
+ : ''
346
+
347
+ // A completion can carry text in blocks this strict filter does not name — the tolerant
348
+ // extractor reads any block with a string `text`. Only consulted once the strict pass
349
+ // found nothing, so the usual path keeps its exact spacing.
350
+ if (output.trim() === '') {
351
+ output = textOf(result.content)
352
+ }
328
353
 
329
354
  const entry = await spectate(spectator, 'ask')(msgs, message, action, i, startedAt)
330
355
  if (ref != null) ref.spectatorEntry = entry
331
356
 
357
+ // An empty completion is a NULL RESULT, and it is diagnosed here rather than blamed on
358
+ // the caller. Both shipped filters return null only for empty input, so letting one run
359
+ // first reported every empty answer as `filter-rejected` — naming the innocent party and,
360
+ // worse, skipping `reportNull`, whose stop reason and output-token count are the only
361
+ // things that say WHY nothing came back (a model that spent its whole budget thinking).
362
+ if (output.trim() === '') {
363
+ throw await nullResult('ask', {
364
+ action, attempt: i, startedAt, refined, msgs, raw: result, useCache,
365
+ })
366
+ }
367
+
332
368
  if (filter != null) {
333
- output = await filter(output ?? '', message)
334
- if (output == null) {
335
- throw new LlmModelError(`filter-rejected:${JSON.stringify(message).substring(0, 50)}...`)
369
+ const produced = output
370
+ const filtered = await filter(produced, message)
371
+ if (filtered == null) {
372
+ // The OUTPUT, not the message envelope: `JSON.stringify(new AIMessage(...))` is 80
373
+ // constant characters of LangChain serialization stub and says nothing at all.
374
+ throw new LlmModelError(`filter-rejected:${produced.substring(0, 200)}`)
336
375
  }
337
- } else if (output == null || output.trim() === '') {
338
- throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
376
+ output = filtered
339
377
  }
340
378
 
341
379
  notifyRef(ref, message)
@@ -345,11 +383,15 @@ export const makeLlmModel = ({
345
383
 
346
384
  talk: async (
347
385
  input,
348
- { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmTalkOptions
386
+ {
387
+ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
388
+ escalation, fatal,
389
+ }: LlmTalkOptions
349
390
  ) => {
350
391
  const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
351
- return withRetry({ retries, outputErrors }, async i => {
352
- const refined = refineModel(i)
392
+ const seed = ladderSeed(escalation)
393
+ return withRetry({ retries, outputErrors, fatal }, async i => {
394
+ const refined = refineModel(seed + i)
353
395
  console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
354
396
  const startedAt = Date.now()
355
397
  let result: AIMessageChunk | null = null
@@ -381,14 +423,18 @@ export const makeLlmModel = ({
381
423
  invoke: async <T>(
382
424
  input: ModelInput,
383
425
  schema: JSONSchemaType<T>,
384
- { temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmInvokeOptions<T>
426
+ {
427
+ temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS,
428
+ skills, escalation, fatal,
429
+ }: LlmInvokeOptions<T>
385
430
  ) => {
386
431
  const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
387
432
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
388
433
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
389
434
 
390
- return withRetry({ retries, outputErrors }, async i => {
391
- const refined = refineModel(i, temperature)
435
+ const seed = ladderSeed(escalation)
436
+ return withRetry({ retries, outputErrors, fatal }, async i => {
437
+ const refined = refineModel(seed + i, temperature)
392
438
  console.log('Use model invoke: ', refined.getName(), refined.lc_kwargs.model)
393
439
  const startedAt = Date.now()
394
440
  const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
@@ -429,14 +475,18 @@ export const makeLlmModel = ({
429
475
  request: async <T>(
430
476
  input: ModelInput,
431
477
  schema: JSONSchemaType<T>,
432
- { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmRequestOptions
478
+ {
479
+ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
480
+ escalation, fatal,
481
+ }: LlmRequestOptions
433
482
  ) => {
434
483
  const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
435
484
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
436
485
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
437
486
 
438
- return withRetry({ retries, outputErrors }, async i => {
439
- const refined = refineModel(i)
487
+ const seed = ladderSeed(escalation)
488
+ return withRetry({ retries, outputErrors, fatal }, async i => {
489
+ const refined = refineModel(seed + i)
440
490
  console.log('Use model request: ', refined.getName(), refined.lc_kwargs.model)
441
491
  const startedAt = Date.now()
442
492
  const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
@@ -5,13 +5,68 @@ import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messa
5
5
  import { ModelProvider, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
6
6
  import type { CacheTtl } from '@owlmeans/llm-common'
7
7
  import type { LlmPlugin } from './types.js'
8
+ import type { ModelConfig } from '../types.js'
8
9
  import { CHARS_PER_TOKEN, MAX_CACHE_BREAKPOINTS, MIN_CACHEABLE_TOKENS } from '../consts.js'
10
+ import { resolveOutputCap } from '../utils/config.js'
9
11
  import { readConfig } from '../utils/config.js'
10
12
  import { escalateMaxTokens, isBadRequest, makeClientOptions } from './utils.js'
11
13
 
12
14
  /** Model-name prefix that supports prompt caching through `cache_control` markers. */
13
15
  const CACHEABLE_PREFIX = 'claude-'
14
16
 
17
+ /**
18
+ * Model families that REJECT the sampling parameters — Claude 4.7 and later, and the whole
19
+ * 5 family. `temperature`, `top_p` and `top_k` were removed there, and sending any of them
20
+ * is a 400, not a silently ignored field. Matched with `startsWith`, so a dated snapshot
21
+ * (`claude-sonnet-5-20260114`) is covered by its base id.
22
+ *
23
+ * This is the Anthropic counterpart of the OpenAI plugin's `RESPONSES_API_PREFIXES`: the
24
+ * older models below the line (`claude-sonnet-4-6`, `claude-haiku-4-5`, and earlier) still
25
+ * accept sampling and still want the deterministic `temperature: 0` default.
26
+ */
27
+ export const NO_SAMPLING_PREFIXES = [
28
+ 'claude-fable-5',
29
+ 'claude-mythos-5',
30
+ 'claude-mythos-preview',
31
+ 'claude-opus-5',
32
+ 'claude-opus-4-8',
33
+ 'claude-opus-4-7',
34
+ 'claude-sonnet-5',
35
+ ]
36
+
37
+ /** Whether this model id rejects `temperature`/`top_p`/`top_k`. */
38
+ export const rejectsSampling = (model: string | undefined): boolean =>
39
+ model != null && NO_SAMPLING_PREFIXES.some(prefix => model.startsWith(prefix))
40
+
41
+ /**
42
+ * Whether the request has to say, on the wire, that the model must not reason.
43
+ *
44
+ * The adaptive family reasons unless told otherwise: an absent `thinking` parameter means
45
+ * "adaptive", and langchain forwards the parameter only when a caller sets it — so a config
46
+ * that asks for no thinking is only honoured if the plugin sends `thinking: disabled` itself.
47
+ * Silent reasoning is what the request pays for twice: its tokens bill as output, and the
48
+ * summarised stream delivers them in bursts minutes apart, which an idle deadline reads as a
49
+ * dead connection and retries from scratch. Older models reason only when asked and get nothing.
50
+ */
51
+ export const suppressesThinking = (config: Pick<ModelConfig, 'model' | 'disableThinking'>): boolean =>
52
+ config.disableThinking === true && rejectsSampling(config.model)
53
+
54
+ /**
55
+ * The smallest output budget an always-reasoning model is given.
56
+ *
57
+ * The same models that took the sampling knobs away also think ADAPTIVELY unless the request
58
+ * turns it off (`disableThinking` → `thinking: disabled`, see `suppressesThinking`), and by
59
+ * default that thinking is not shown — it arrives as thinking blocks with empty text. Reasoning is billed against the same `max_tokens` as the answer, so a budget
60
+ * sized for the answer alone can be spent entirely on thinking: the response is a well-formed
61
+ * completion carrying no text at all, `stop_reason: "max_tokens"`, and every retry at the same
62
+ * budget draws from the same distribution.
63
+ *
64
+ * The floor buys room for the reasoning AND the answer. It is a floor, not an override — a preset
65
+ * asking for more keeps it — and it is clamped to what the provider accepts, so it can never turn
66
+ * a retryable empty answer into a 400.
67
+ */
68
+ export const ADAPTIVE_MIN_MAX_TOKENS = 32_000
69
+
15
70
  export const ANTHROPIC_FAMILY = 'anthropic'
16
71
 
17
72
  type ContentBlock = Record<string, unknown>
@@ -91,19 +146,38 @@ export const anthropicPlugin: LlmPlugin = {
91
146
  */
92
147
  toolChoice: (toolName: string): unknown => ({ type: 'tool', name: toolName }),
93
148
 
149
+ suppressesThinking: config => suppressesThinking(config),
150
+
94
151
  build: ({ config, secret, callbacks }) => {
95
- const model = config.model ??= 'claude-haiku-4-5-20251001'
152
+ const model = config.model ??= 'claude-haiku-4-5'
153
+ // Claude 4.7+ took the sampling knobs away: not "ignored", a 400. A configured
154
+ // `temperature` on such a model is a preset bug, and dropping it here is the only
155
+ // reading that keeps the call alive — there is nothing to translate it into.
156
+ const sampling = rejectsSampling(model)
157
+ ? {}
158
+ : {
159
+ // Neither knob set → pin temperature to 0 for determinism.
160
+ ...(config.temperature == null && config.topP == null ? { temperature: 0 } : {}),
161
+ ...(config.temperature != null ? { temperature: config.temperature } : {}),
162
+ ...(config.topP != null && config.temperature == null ? { topP: config.topP } : {}),
163
+ }
164
+ // Room for the reasoning these models always do, and for the answer after it.
165
+ const requested = config.maxTokens ?? 4096
166
+ const maxTokens = rejectsSampling(model)
167
+ ? Math.min(Math.max(requested, ADAPTIVE_MIN_MAX_TOKENS), resolveOutputCap(config))
168
+ : requested
169
+
96
170
  const cfg = {
97
171
  model,
98
172
  apiKey: secret,
99
- // Neither knob set → pin temperature to 0 for determinism.
100
- ...(config.temperature == null && config.topP == null ? { temperature: 0 } : {}),
101
- maxTokens: config.maxTokens ?? 4096,
173
+ maxTokens,
102
174
  maxRetries: 5,
103
175
  metadata: { config },
104
176
  callbacks,
105
- ...(config.temperature != null ? { temperature: config.temperature } : {}),
106
- ...(config.topP != null && config.temperature == null ? { topP: config.topP } : {}),
177
+ ...sampling,
178
+ ...(suppressesThinking({ model, disableThinking: config.disableThinking })
179
+ ? { thinking: { type: 'disabled' as const } }
180
+ : {}),
107
181
  ...makeClientOptions({ headers: config.headers }),
108
182
  }
109
183
  // Anthropic rejects temperature and top_p together.
@@ -118,6 +192,15 @@ export const anthropicPlugin: LlmPlugin = {
118
192
  const model = base as ChatAnthropic
119
193
  const currentTemperature = temperature ?? model.temperature ?? 0
120
194
  const maxTokens = escalateMaxTokens(model.maxTokens, attempt, maxOutputCap)
195
+ // `lc_kwargs` carries whatever `build` put there, so a no-sampling model arrives clean;
196
+ // what has to be suppressed is the escalator's own re-application of a temperature.
197
+ if (rejectsSampling(model.modelName ?? model.model)) {
198
+ const cfg = { ...(model.lc_kwargs as Partial<ChatAnthropic>), maxTokens }
199
+ delete cfg.temperature
200
+ delete cfg.topP
201
+
202
+ return new ChatAnthropic(cfg as Partial<ChatAnthropic>)
203
+ }
121
204
  const cfg: Partial<ChatAnthropic> = {
122
205
  ...(model.lc_kwargs as Partial<ChatAnthropic>), temperature: currentTemperature, maxTokens,
123
206
  }
@@ -5,8 +5,20 @@ import type { LlmPlugin, LlmRefineParams } from './types.js'
5
5
  import type { ModelConfig } from '../types.js'
6
6
  import { escalateMaxTokens, isBadRequest, makeConfiguration } from './utils.js'
7
7
 
8
- /** Model families served through OpenAI's Responses API rather than chat completions. */
9
- const RESPONSES_API_PREFIXES = ['gpt-5', 'codex-']
8
+ /**
9
+ * Model families served through OpenAI's Responses API rather than chat completions. That
10
+ * endpoint REJECTS `temperature`/`top_p` — a 400 naming the parameter, not a silently
11
+ * ignored field. Matched with `startsWith`, so a dated snapshot (`gpt-5.6-terra-2026-08`)
12
+ * is covered by its base id.
13
+ *
14
+ * This is the OpenAI counterpart of the anthropic plugin's `NO_SAMPLING_PREFIXES`, and it
15
+ * gates BOTH hooks for the same reason: see `refine`.
16
+ */
17
+ export const RESPONSES_API_PREFIXES = ['gpt-5', 'codex-']
18
+
19
+ /** Whether this model id goes through the Responses API and therefore rejects sampling. */
20
+ export const usesResponsesApi = (model: string | undefined): boolean =>
21
+ model != null && RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))
10
22
 
11
23
  export const OPENAI_FAMILY = 'openai'
12
24
 
@@ -52,6 +64,8 @@ export const openAiFamily = {
52
64
  const baseKwargs = model.lc_kwargs as ConstructorParameters<typeof ChatOpenAI>[0] & {
53
65
  modelKwargs?: { reasoning?: { max_tokens?: number } } & Record<string, unknown>
54
66
  }
67
+ const responsesApi = usesResponsesApi(model.model ?? baseKwargs.model)
68
+ || baseKwargs.useResponsesApi === true
55
69
  // The dominant cause of an empty response is a reasoning model spending the whole
56
70
  // budget on hidden thinking (finish_reason=length, empty content). The retry already
57
71
  // raises maxTokens; ALSO shrink the absolute reasoning cap so the extra budget becomes
@@ -65,6 +79,22 @@ export const openAiFamily = {
65
79
  }
66
80
  : baseKwargs.modelKwargs
67
81
 
82
+ // `build` hands a Responses-API model over without sampling knobs, but EVERY call is
83
+ // made on the instance `refine` returns — attempt 0 included — so re-applying a
84
+ // temperature here puts it on the wire for every single request, not just a retry.
85
+ // Suppressing it in one hook and restoring it in the other ships the parameter anyway.
86
+ if (responsesApi) {
87
+ const cfg = {
88
+ ...baseKwargs,
89
+ maxTokens,
90
+ ...(modelKwargs != null ? { modelKwargs } : {}),
91
+ }
92
+ delete cfg.temperature
93
+ delete cfg.topP
94
+
95
+ return new ChatOpenAI(cfg)
96
+ }
97
+
68
98
  return new ChatOpenAI({
69
99
  ...baseKwargs,
70
100
  temperature: currentTemperature,
@@ -104,7 +134,7 @@ export const openAiPlugin: LlmPlugin = {
104
134
  const modelKwargs = { prompt_cache_key: config.cacheKey ?? alias }
105
135
 
106
136
  // The Responses API models reject `temperature`/`topP`.
107
- if (RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))) {
137
+ if (usesResponsesApi(model)) {
108
138
  return new ChatOpenAI({
109
139
  model,
110
140
  apiKey: secret,
@@ -93,6 +93,14 @@ export interface LlmPlugin {
93
93
  */
94
94
  refine: (params: LlmRefineParams) => BaseChatModel
95
95
 
96
+ /**
97
+ * Does this plugin turn the model's reasoning off NATIVELY when the config asks for it
98
+ * (`ModelConfig.disableThinking`)? When it does, the service must not also inject the
99
+ * `/no_think` prompt directive — that is a soft switch for models with no request-level
100
+ * control, and on a provider that has one it is nothing but text in the prompt.
101
+ */
102
+ suppressesThinking?: (config: Pick<ModelConfig, 'model' | 'disableThinking'>) => boolean
103
+
96
104
  /** How this provider should be asked for schema-conforming output. */
97
105
  structuredMode: (config: ModelConfig) => StructuredMode
98
106
 
@@ -105,6 +105,10 @@ export const promptServiceApi = (
105
105
 
106
106
  compose: async (input, messages, params): Promise<PromptResult> => {
107
107
  const sections = new Map<PromptBlock, string[]>()
108
+ // Scoped to this composition, never to the service: a claim is about who renders a
109
+ // thing in ONE prompt, and carrying it across calls would silently drop the content
110
+ // from every later prompt that shares the service.
111
+ const claimed = new Set<string>()
108
112
  const ctx: PromptContext = {
109
113
  ...params,
110
114
  input,
@@ -122,6 +126,14 @@ export const promptServiceApi = (
122
126
  }
123
127
  },
124
128
  resolve: aliases => self().resolve(aliases),
129
+ claim: key => {
130
+ if (claimed.has(key)) {
131
+ return false
132
+ }
133
+ claimed.add(key)
134
+
135
+ return true
136
+ },
125
137
  }
126
138
 
127
139
  // Two passes, not one: every static contribution must be in place before a plugin
@@ -36,6 +36,16 @@ export interface PromptComposeParams {
36
36
  cacheMax?: number
37
37
  /** File access a plugin may use to resolve knowledge from disk. */
38
38
  files?: FileProviderRef
39
+ /**
40
+ * A cheap model a plugin may spend ONE call on while composing — picking which of a
41
+ * hundred candidate skills a request is actually about, classifying an ask. Resolved
42
+ * lazily and allowed to yield `undefined`: no cheap tier is configured on most
43
+ * deployments, and a plugin that cannot get one must degrade rather than fail.
44
+ *
45
+ * Whatever it returns must not change what lands in a CACHED block — a model's answer
46
+ * is not reproducible byte-for-byte, so it belongs in `Packages` or `Context`.
47
+ */
48
+ utility?: () => BaseChatModel | undefined
39
49
  }
40
50
 
41
51
  /** What a prompt plugin sees and may contribute to. */
@@ -47,6 +57,16 @@ export interface PromptContext extends PromptComposeParams {
47
57
  add: (block: PromptBlock, text: string) => void
48
58
  /** Resolve skill aliases through the registry, following `requires`. */
49
59
  resolve: (aliases: readonly string[]) => SkillDefinition[]
60
+ /**
61
+ * Take exclusive ownership of `key` for THIS composition: the first caller gets `true`,
62
+ * every later one `false`. Two plugins that can each render the same skill — a static
63
+ * catalogue and a detector — would otherwise emit it twice, which costs tokens and
64
+ * tells the model the same thing in two voices.
65
+ *
66
+ * The claim set is per `compose` call and consulted by nobody else, so a composition
67
+ * where no plugin claims renders exactly the bytes it rendered before this seam existed.
68
+ */
69
+ claim: (key: string) => boolean
50
70
  }
51
71
 
52
72
  /**
package/src/service.ts CHANGED
@@ -38,11 +38,39 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
38
38
  })
39
39
  }
40
40
 
41
+ /** A named alias's config, ready to be layered under something else. */
42
+ const presetOf = (
43
+ models: ModelConfig[], name: string | undefined
44
+ ): Partial<ModelConfig> => {
45
+ if (name == null) return {}
46
+ const { alias: _alias, ...rest } = { ...(models.find(m => m.alias === name) ?? {}) }
47
+ return rest
48
+ }
49
+
50
+ /**
51
+ * Drop keys that are present but `undefined` — they must not shadow a layer below.
52
+ * `mergeOverride` in the execution layer strips these already; a hand-built override
53
+ * need not.
54
+ */
55
+ const defined = (config: Partial<ModelConfig>): Partial<ModelConfig> =>
56
+ Object.fromEntries(
57
+ Object.entries(config).filter(([, value]) => value !== undefined)
58
+ ) as Partial<ModelConfig>
59
+
41
60
  /**
42
61
  * Resolve `alias` → config (inheriting a `preset`, applying `override`) and build it.
43
62
  * A declared `fallback` is built as well and attached to the primary as a
44
63
  * non-enumerable `__fallbackModel`, which the model's retry escalator reads. The
45
64
  * fallback spec is merged OVER the primary config, so it inherits secret/headers.
65
+ *
66
+ * Four layers, lowest first — the alias's own preset, the alias, the override's preset,
67
+ * the override. A `preset` is a BASE that its referent refines, so it has to sit under
68
+ * the config naming it; the previous order assigned it last, which meant a role
69
+ * declaring `preset:` silently discarded both its own fields and the caller's override —
70
+ * effort-tier token caps and `temperatureFactory`'s temperature among them. An override
71
+ * naming a preset (how the execution layer delivers a `modelOverrides` string pin) still
72
+ * outranks the alias, because picking a different model is a stronger statement than the
73
+ * role's default; explicit override fields stay on top of everything.
46
74
  */
47
75
  const createModel = (alias: string, override: Partial<ModelConfig> = {}): BaseChatModel => {
48
76
  const models = options.models()
@@ -50,17 +78,35 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
50
78
  if (baseConfig == null) {
51
79
  throw new LlmMissconfiguredError(alias)
52
80
  }
53
- const config: ModelConfig = { ...baseConfig, ...override }
81
+ const config: ModelConfig = {
82
+ ...presetOf(models, baseConfig.preset),
83
+ ...defined(baseConfig),
84
+ ...presetOf(models, override.preset),
85
+ ...defined(override),
86
+ alias: baseConfig.alias,
87
+ }
54
88
  // The service-wide idle deadline is a floor, not an override: a preset that states its
55
89
  // own `streamTimeout` knows something specific about that model and keeps it.
56
90
  config.streamTimeout ??= options.streamTimeout
57
- const preset: Partial<ModelConfig> = config.preset != null
58
- ? { ...(models.find(m => m.alias === config.preset) ?? {}) }
59
- : {}
60
- if (preset.alias != null) {
61
- delete preset.alias
91
+
92
+ // What the provider accepts bounds what we may ask for. A preset that over-declares is
93
+ // corrected here rather than at the provider, where it surfaces as a fatal 400 on the
94
+ // one call that finally escalated far enough to exceed the limit.
95
+ if (config.maxOutput != null && config.maxOutput > 0) {
96
+ if (config.maxTokensCap != null && config.maxTokensCap > config.maxOutput) {
97
+ console.warn(
98
+ `Model "${alias}" declares maxTokensCap ${config.maxTokensCap} above the provider's`
99
+ + ` maxOutput ${config.maxOutput}; the escalator will stop at ${config.maxOutput}.`
100
+ )
101
+ }
102
+ if (config.maxTokens != null && config.maxTokens > config.maxOutput) {
103
+ console.warn(
104
+ `Model "${alias}" declares maxTokens ${config.maxTokens} above the provider's`
105
+ + ` maxOutput ${config.maxOutput}; clamping.`
106
+ )
107
+ config.maxTokens = config.maxOutput
108
+ }
62
109
  }
63
- Object.assign(config, preset)
64
110
 
65
111
  const { fallback, ...primaryConfig } = config
66
112
  const primary = buildModel(alias, primaryConfig)