@owlmeans/llm 0.1.18-rc.2 → 0.1.18-rc.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +2 -2
  2. package/agent-meta/manifest.json +2 -2
  3. package/agent-meta/skills/llm/SKILL.md +239 -116
  4. package/agent-meta/skills/llm-prompt-caching/SKILL.md +37 -3
  5. package/build/execution/service.d.ts.map +1 -1
  6. package/build/execution/service.js +46 -11
  7. package/build/execution/service.js.map +1 -1
  8. package/build/execution/types.d.ts +72 -12
  9. package/build/execution/types.d.ts.map +1 -1
  10. package/build/execution/utils.d.ts.map +1 -1
  11. package/build/execution/utils.js +15 -9
  12. package/build/execution/utils.js.map +1 -1
  13. package/build/helpers/retry.d.ts +14 -0
  14. package/build/helpers/retry.d.ts.map +1 -1
  15. package/build/helpers/retry.js +14 -0
  16. package/build/helpers/retry.js.map +1 -1
  17. package/build/helpers/spectate.d.ts.map +1 -1
  18. package/build/helpers/spectate.js +12 -5
  19. package/build/helpers/spectate.js.map +1 -1
  20. package/build/index.d.ts +2 -2
  21. package/build/index.d.ts.map +1 -1
  22. package/build/index.js +2 -2
  23. package/build/index.js.map +1 -1
  24. package/build/model.d.ts +1 -1
  25. package/build/model.d.ts.map +1 -1
  26. package/build/model.js +60 -28
  27. package/build/model.js.map +1 -1
  28. package/build/plugins/anthropic.d.ts +40 -0
  29. package/build/plugins/anthropic.d.ts.map +1 -1
  30. package/build/plugins/anthropic.js +79 -6
  31. package/build/plugins/anthropic.js.map +1 -1
  32. package/build/plugins/openai.d.ts +12 -0
  33. package/build/plugins/openai.d.ts.map +1 -1
  34. package/build/plugins/openai.js +29 -3
  35. package/build/plugins/openai.js.map +1 -1
  36. package/build/plugins/types.d.ts +7 -0
  37. package/build/plugins/types.d.ts.map +1 -1
  38. package/build/prompt/service.d.ts.map +1 -1
  39. package/build/prompt/service.js +11 -0
  40. package/build/prompt/service.js.map +1 -1
  41. package/build/prompt/types.d.ts +20 -0
  42. package/build/prompt/types.d.ts.map +1 -1
  43. package/build/service.d.ts.map +1 -1
  44. package/build/service.js +42 -7
  45. package/build/service.js.map +1 -1
  46. package/build/types.d.ts +71 -1
  47. package/build/types.d.ts.map +1 -1
  48. package/build/utils/config.d.ts +11 -0
  49. package/build/utils/config.d.ts.map +1 -1
  50. package/build/utils/config.js +21 -1
  51. package/build/utils/config.js.map +1 -1
  52. package/build/utils/null-report.d.ts.map +1 -1
  53. package/build/utils/null-report.js +9 -1
  54. package/build/utils/null-report.js.map +1 -1
  55. package/package.json +13 -13
  56. package/src/execution/service.ts +46 -11
  57. package/src/execution/types.ts +77 -13
  58. package/src/execution/utils.ts +16 -9
  59. package/src/helpers/retry.ts +16 -0
  60. package/src/helpers/spectate.ts +14 -5
  61. package/src/index.ts +4 -2
  62. package/src/model.ts +77 -27
  63. package/src/plugins/anthropic.ts +89 -6
  64. package/src/plugins/openai.ts +33 -3
  65. package/src/plugins/types.ts +8 -0
  66. package/src/prompt/service.ts +12 -0
  67. package/src/prompt/types.ts +20 -0
  68. package/src/service.ts +53 -7
  69. package/src/types.ts +71 -1
  70. package/src/utils/config.ts +23 -1
  71. package/src/utils/null-report.ts +9 -1
  72. package/tests/context.ts +3 -0
  73. package/tests/execution.spec.ts +148 -13
  74. package/tests/helpers.spec.ts +4 -1
  75. package/tests/plugins.spec.ts +211 -0
  76. package/tests/prompt.spec.ts +43 -0
package/src/types.ts CHANGED
@@ -63,6 +63,12 @@ export interface LlmModelOptions extends LlmLogging {
63
63
  prompts?: () => PromptService
64
64
  /** File access offered to prompt plugins that resolve knowledge from disk. */
65
65
  files?: FileProviderRef
66
+ /**
67
+ * Cheap model offered to prompt plugins for a single side call while composing —
68
+ * normally `() => executions().utility(exec)`. Omitted, plugins that would use one
69
+ * fall back to whatever they can decide without a model.
70
+ */
71
+ utility?: () => BaseChatModel | undefined
66
72
  }
67
73
 
68
74
  export type ModelMessage = BaseMessage | MessageFieldWithRole
@@ -89,6 +95,27 @@ export interface LlmCallOptions {
89
95
  * should be part of the shared prefix.
90
96
  */
91
97
  skills?: string[]
98
+ /**
99
+ * How many failures of this SAME piece of work the caller has already observed — the
100
+ * per-call escalator starts that far up its ladder instead of at the bottom.
101
+ *
102
+ * A caller that validates the OUTPUT (a diff that must apply, a file that must not come
103
+ * back truncated) runs its own retry loop around whole calls, and every one of those
104
+ * calls used to start at attempt 0: same model, same output budget, same answer. Passing
105
+ * the outer attempt here advances both rungs the escalator owns — the `maxTokens`
106
+ * doubling and the {@link FALLBACK_AFTER_ATTEMPTS} switch to `ModelConfig.fallback` — so
107
+ * a repeatedly-rejected call actually reaches the stronger model.
108
+ *
109
+ * Clamped to `retries - 1`; it moves the STARTING rung only and never changes how many
110
+ * attempts this call makes. Unrelated to `ExecutionService.escalate`, which raises the
111
+ * effort tier of an execution before any model is resolved.
112
+ */
113
+ escalation?: number
114
+ /**
115
+ * Abort THIS call's retry loop for an error no retry can fix. Consulted before the
116
+ * globally registered resolvers and the provider plugin's `isFatal`.
117
+ */
118
+ fatal?: (e: unknown) => Error | null
92
119
  }
93
120
 
94
121
  export interface LlmAskOptions extends LlmCallOptions {
@@ -151,20 +178,63 @@ export interface TemperatureFactory {
151
178
  export interface ModelConfig {
152
179
  provider?: ModelProvider | string
153
180
  secret?: string
181
+ /**
182
+ * Which delegate transport answers this model's calls.
183
+ *
184
+ * Only meaningful for {@link ModelProvider.Delegated}: the key the application seated a
185
+ * transport under, so one deployment can hold many at once — one per connected agent — and a
186
+ * config names the one that belongs to its run.
187
+ */
188
+ delegate?: string
154
189
  alias: string
155
190
  /** Inherit every field of another alias in the same config list. */
156
191
  preset?: string
157
192
  model?: string
158
193
  temperature?: number
159
- /** Initial output-token budget per request (`max_tokens`). */
194
+ /**
195
+ * Initial output-token budget per request (`max_tokens`) — what THIS deployment asks
196
+ * for first, not what the model can do. Clamped to {@link ModelConfig.maxOutput}.
197
+ */
160
198
  maxTokens?: number
161
199
  /**
162
200
  * Hard ceiling for output tokens used by the retry escalator. Each retry doubles
163
201
  * `maxTokens` toward this cap; without it the escalator uses
164
202
  * {@link DEFAULT_MAX_OUTPUT_CAP}, which exceeds many models' real per-request output
165
203
  * limit and turns retries into 400 "max_tokens exceeds model limit" errors.
204
+ *
205
+ * This is the budget the PRESET chooses; {@link ModelConfig.maxOutput} is what the
206
+ * provider allows. The effective ceiling is the smaller of the two, so a preset that
207
+ * over-declares is corrected at refine time rather than at the provider.
166
208
  */
167
209
  maxTokensCap?: number
210
+ /**
211
+ * Total context window the model accepts — input plus output. Informational: it is
212
+ * never sent to the provider and the runtime cannot enforce it (the input size is not
213
+ * known at config time). It documents the model and lets the preset test catch a
214
+ * `maxOutput` that could not possibly fit.
215
+ */
216
+ contextWindow?: number
217
+ /**
218
+ * What the PROVIDER accepts as output in a single request. Unlike `maxTokens` and
219
+ * `maxTokensCap` — both deployment choices — this is a property of the model behind
220
+ * this alias, and for an aggregated model it is the limit of the `inferenceProvider`
221
+ * actually pinned here, which is often well below what the model can do elsewhere.
222
+ *
223
+ * Hard ceiling: `maxTokens` is clamped to it at build time and the escalator's cap is
224
+ * `min(maxTokensCap, maxOutput)`.
225
+ *
226
+ * A `fallback` that changes `model` MUST redeclare this (and `contextWindow`), because
227
+ * the fallback config inherits every field the patch does not name — otherwise the
228
+ * escalator would size the fallback by the primary's capability.
229
+ */
230
+ maxOutput?: number
231
+ /**
232
+ * The context window is SHARED between input and output rather than being an input
233
+ * allowance with a separate output limit (MiniMax M2.x, gpt-oss). Nothing enforces it
234
+ * at runtime; it marks the entry so `maxOutput` is read as "spends the same budget the
235
+ * prompt spends" and keeps presets honest about leaving room for input.
236
+ */
237
+ combinedWindow?: boolean
168
238
  topP?: number
169
239
  baseUrl?: string
170
240
  organization?: string
@@ -1,5 +1,5 @@
1
1
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
- import { MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
2
+ import { DEFAULT_MAX_OUTPUT_CAP, MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
3
3
  import type { ModelConfig } from '../types.js'
4
4
 
5
5
  /**
@@ -17,3 +17,25 @@ export const readConfig = (model: BaseChatModel): Partial<ModelConfig> => {
17
17
  /** Per-model idle stream timeout (ms), falling back to the package default. */
18
18
  export const idleTimeout = (config: Partial<ModelConfig>): number =>
19
19
  config.streamTimeout ?? MODEL_STREAM_TIMEOUT_MS
20
+
21
+ /**
22
+ * The output ceiling the retry escalator may climb to.
23
+ *
24
+ * Two different things claim to bound the output, and only one of them is a fact:
25
+ * `maxTokensCap` is what this deployment budgeted, `maxOutput` is what the provider will
26
+ * accept. Taking the declared cap alone let a preset out-declare its model and turned
27
+ * every escalation into a 400; taking the capability alone would ignore a deliberately
28
+ * frugal budget. So the declared value chooses the ceiling and the capability trims it,
29
+ * and only when neither is stated does {@link DEFAULT_MAX_OUTPUT_CAP} apply.
30
+ */
31
+ export const resolveOutputCap = (config: Partial<ModelConfig>): number => {
32
+ const declared = typeof config.maxTokensCap === 'number' && config.maxTokensCap > 0
33
+ ? config.maxTokensCap
34
+ : undefined
35
+ const capability = typeof config.maxOutput === 'number' && config.maxOutput > 0
36
+ ? config.maxOutput
37
+ : undefined
38
+ const cap = declared ?? capability ?? DEFAULT_MAX_OUTPUT_CAP
39
+
40
+ return capability != null ? Math.min(cap, capability) : cap
41
+ }
@@ -45,6 +45,7 @@ export const buildNullReport = (p: NullReportParams): NullCapture => {
45
45
  finish_reason?: string
46
46
  usage?: { prompt_tokens?: number; completion_tokens?: number; reasoning_tokens?: number }
47
47
  } | undefined
48
+ const stopReason = (raw?.additional_kwargs as { stop_reason?: string } | undefined)?.stop_reason
48
49
  const usageMeta = raw?.usage_metadata as { input_tokens?: number; output_tokens?: number } | undefined
49
50
  const toolCalls = (raw as unknown as { tool_calls?: unknown[] } | null)?.tool_calls
50
51
 
@@ -81,7 +82,14 @@ export const buildNullReport = (p: NullReportParams): NullCapture => {
81
82
  tool_calls: toolCalls,
82
83
  } : null,
83
84
  diagnostics: {
84
- finishReason: responseMeta?.finish_reason,
85
+ // Two providers, two places. OpenAI-compatible APIs put it on `response_metadata`;
86
+ // Anthropic never does — it arrives on the `message_delta` event and langchain spreads it
87
+ // into `additional_kwargs.stop_reason`. Reading only the first printed `undefined` for
88
+ // every Anthropic null, hiding the `max_tokens` that explains most of them.
89
+ finishReason: responseMeta?.finish_reason ?? stopReason,
90
+ /** No text block at all — the shape of a completion that was all reasoning. */
91
+ thinkingOnly: Array.isArray(raw?.content) && raw.content.length > 0
92
+ && !raw.content.some(part => (part as { type?: string }).type === 'text'),
85
93
  inputTokens: usageMeta?.input_tokens ?? responseMeta?.usage?.prompt_tokens,
86
94
  outputTokens: usageMeta?.output_tokens ?? responseMeta?.usage?.completion_tokens,
87
95
  reasoningTokens: responseMeta?.usage?.reasoning_tokens,
package/tests/context.ts CHANGED
@@ -16,6 +16,8 @@ export const gates = makeGates({
16
16
  export const Role = {
17
17
  Analyst: 'analyst',
18
18
  Picker: 'picker',
19
+ /** The conventional cheap tier — same value as `UTILITY_ROLE`. */
20
+ Utility: 'utility',
19
21
  } as const
20
22
 
21
23
  /**
@@ -60,6 +62,7 @@ export const anthropicConfigs = (): ModelConfig[] => [
60
62
  */
61
63
  export const offlineConfigs = (): ModelConfig[] => [
62
64
  { alias: Role.Analyst, provider: ModelProvider.OpenAI, model: 'gpt-4.1-mini', secret: 'sk-test' },
65
+ { alias: Role.Utility, provider: ModelProvider.OpenAI, model: 'gpt-4.1-nano', secret: 'sk-test' },
63
66
  {
64
67
  alias: Role.Picker, provider: ModelProvider.Compatible, model: 'some/model', secret: 'sk-test',
65
68
  baseUrl: 'https://openrouter.ai/api/v1',
@@ -1,5 +1,5 @@
1
1
  import { beforeEach, describe, expect, test } from 'bun:test'
2
- import { ExecutionEffort, ExecutionLevel } from '@owlmeans/llm-common'
2
+ import { ExecutionEffort, ExecutionLevel, UTILITY_ROLE } from '@owlmeans/llm-common'
3
3
  import type { ExecutionState, ModelConfigPatch, TaskExecutionState } from '@owlmeans/llm-common'
4
4
  import { DEFAULT_EFFORT, EFFORT_TABLE, makeExecutionService, makeLlmService } from '@owlmeans/llm'
5
5
  import type { ExecutionService, ProjectExecution, TaskExecution } from '@owlmeans/llm'
@@ -135,6 +135,65 @@ describe('@owlmeans/llm — model policy resolution', () => {
135
135
  })
136
136
  })
137
137
 
138
+ describe('@owlmeans/llm — utility tier', () => {
139
+ test('the conventional role is resolved at the cheapest tier', () => {
140
+ service.utility(root)
141
+ expect(resolved.at(-1))
142
+ .toEqual({ alias: UTILITY_ROLE, override: EFFORT_TABLE[ExecutionEffort.Economy] })
143
+ })
144
+
145
+ // The whole point of routing it through `model`: a deployment remaps or pins the cheap
146
+ // tier with the levers it already uses for every other role.
147
+ test('a role override remaps which model the utility tier resolves', () => {
148
+ const remapped = service.escalate(root, { roleOverrides: { [UTILITY_ROLE]: Role.Picker } })
149
+ service.utility(remapped)
150
+ expect(resolved.at(-1)?.alias).toBe(Role.Picker)
151
+ })
152
+
153
+ test('a policy utility role replaces the conventional one, and is still remapped', () => {
154
+ const named = service.root({
155
+ models: makeService(),
156
+ policy: { effort: DEFAULT_EFFORT, utilityRole: Role.Analyst },
157
+ purpose: { type: 'spec' },
158
+ })
159
+ service.utility(named)
160
+ expect(resolved.at(-1)?.alias).toBe(Role.Analyst)
161
+
162
+ service.utility(service.escalate(named, { roleOverrides: { [Role.Analyst]: Role.Picker } }))
163
+ expect(resolved.at(-1)?.alias).toBe(Role.Picker)
164
+ })
165
+
166
+ test('a model override pins the utility role like any other', () => {
167
+ const pinned = service.escalate(root, { modelOverrides: { [UTILITY_ROLE]: { maxTokens: 512 } } })
168
+ service.utility(pinned)
169
+ expect(resolved.at(-1)?.override)
170
+ .toEqual({ ...EFFORT_TABLE[ExecutionEffort.Economy], maxTokens: 512 })
171
+ })
172
+
173
+ // The floor is local — asking for a cheap side model must not quietly downgrade the
174
+ // execution that the real work still runs on.
175
+ test('the economy floor does not touch the execution it was asked on', () => {
176
+ const high = service.escalate(root, { effort: ExecutionEffort.Max })
177
+ service.utility(high)
178
+ expect(high.policy.effort).toBe(ExecutionEffort.Max)
179
+
180
+ service.model(high, Role.Analyst)
181
+ expect(resolved.at(-1)?.override).toEqual(EFFORT_TABLE[ExecutionEffort.Max])
182
+ })
183
+
184
+ test('a named utility role survives refinement of the branch', () => {
185
+ const named = service.root({
186
+ models: makeService(),
187
+ policy: { effort: DEFAULT_EFFORT, utilityRole: Role.Analyst },
188
+ purpose: { type: 'spec' },
189
+ })
190
+ const task = service.forTask(named, { effort: ExecutionEffort.High })
191
+ expect(task.policy.utilityRole).toBe(Role.Analyst)
192
+ service.utility(task)
193
+ expect(resolved.at(-1)?.alias).toBe(Role.Analyst)
194
+ })
195
+ })
196
+
138
197
  describe('@owlmeans/llm — snapshot and restore', () => {
139
198
  test('a snapshot carries the state and none of the collaborators', () => {
140
199
  const state = service.snapshot(root) as ExecutionState & Record<string, unknown>
@@ -222,21 +281,97 @@ describe('@owlmeans/llm — prompt policy accumulation', () => {
222
281
  })
223
282
  })
224
283
 
225
- describe('@owlmeans/llm — resilience plugin seam', () => {
226
- test('checkpoint is a no-op until a plugin is registered', async () => {
227
- await expect(service.checkpoint(root, 'key')).resolves.toBeUndefined()
284
+ describe('@owlmeans/llm — plugin registration', () => {
285
+ test('seats a plugin by alias, so a layer wired twice answers once', async () => {
286
+ // Mixins compose. A plugin registered twice would answer twice, and since the first usable
287
+ // answer wins, the duplicate is silent rather than loud.
288
+ const asked: string[] = []
289
+ service.use({ alias: 'files', advise: async () => { asked.push('first'); return null } })
290
+ service.use({ alias: 'files', advise: async () => { asked.push('second'); return 'answer' } })
291
+
292
+ await service.advise(root, { kind: 'files', task: 'x' })
293
+
294
+ expect(asked).toEqual(['second'])
228
295
  })
229
296
 
230
- test('a registered plugin receives the JSON-safe state and the execution', async () => {
231
- const seen: Array<{ state: ExecutionState, key?: string }> = []
232
- service.use({ onCheckpoint: async (state, _exec, key) => { seen.push({ state, key }) } })
297
+ test('a plugin with no alias is simply appended', async () => {
298
+ const asked: string[] = []
299
+ service.use({ advise: async () => { asked.push('a'); return null } })
300
+ service.use({ advise: async () => { asked.push('b'); return null } })
233
301
 
234
- const task = service.forTask(root, { phase: 'draft' })
235
- await service.checkpoint(task, 'project-1')
302
+ await service.advise(root, { kind: 'files', task: 'x' })
303
+
304
+ expect(asked).toEqual(['a', 'b'])
305
+ })
306
+ })
307
+
308
+ describe('@owlmeans/llm — advice seam', () => {
309
+ test('advise resolves to null when no plugin is registered', async () => {
310
+ await expect(service.advise(root, { kind: 'files', task: 'x' })).resolves.toBeNull()
311
+ })
312
+
313
+ test('an advisor receives the execution and the request', async () => {
314
+ const seen: Array<{ kind: string, task: string, level: string }> = []
315
+ service.use({
316
+ advise: async (exec, request) => {
317
+ seen.push({ kind: request.kind, task: request.task, level: exec.level })
318
+ return 'the card'
319
+ },
320
+ })
321
+
322
+ const helper = service.forHelper(root, { role: Role.Analyst })
323
+ await expect(service.advise(helper, { kind: 'files', task: 'implement X' })).resolves.toBe('the card')
324
+ expect(seen).toEqual([{ kind: 'files', task: 'implement X', level: ExecutionLevel.Helper }])
325
+ })
326
+
327
+ test('the first usable answer wins and later advisors are not consulted', async () => {
328
+ let secondCalled = false
329
+ service.use({ advise: async () => null })
330
+ service.use({ advise: async () => ' ' })
331
+ service.use({ advise: async () => 'first real' })
332
+ service.use({ advise: async () => { secondCalled = true; return 'second' } })
333
+
334
+ await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBe('first real')
335
+ expect(secondCalled).toBe(false)
336
+ })
337
+
338
+ test('a throwing advisor is skipped rather than failing the work', async () => {
339
+ service.use({ advise: async () => { throw new Error('advisor down') } })
340
+ service.use({ advise: async () => 'survived' })
341
+
342
+ await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBe('survived')
343
+ })
344
+
345
+ test('a plugin without advise is ignored', async () => {
346
+ service.use({ onCheckpoint: async () => {} })
347
+ await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBeNull()
348
+ })
349
+ })
350
+
351
+ describe('@owlmeans/llm — per-helper output sizing', () => {
352
+ test('output becomes the resolved model\'s initial maxTokens', () => {
353
+ service.forHelper(root, { role: Role.Analyst, output: 24000 })
354
+
355
+ const call = resolved.at(-1)!
356
+ expect(call.override?.maxTokens).toBe(24000)
357
+ })
358
+
359
+ test('output is not carried onto the helper as a field', () => {
360
+ const helper = service.forHelper(root, { role: Role.Analyst, output: 24000 })
361
+ expect((helper as unknown as { output?: unknown }).output).toBeUndefined()
362
+ })
363
+
364
+ test('a temperature refinement keeps the helper\'s output budget', () => {
365
+ const helper = service.forHelper(root, { role: Role.Analyst, output: 24000 })
366
+ helper.temperatureFactory(0.7)
367
+
368
+ const call = resolved.at(-1)!
369
+ expect(call.override?.maxTokens).toBe(24000)
370
+ expect(call.override?.temperature).toBe(0.7)
371
+ })
236
372
 
237
- expect(seen).toHaveLength(1)
238
- expect(seen[0]!.key).toBe('project-1')
239
- expect((seen[0]!.state as TaskExecutionState).phase).toBe('draft')
240
- expect((seen[0]!.state as unknown as { models?: unknown }).models).toBeUndefined()
373
+ test('without output the override carries no token sizing', () => {
374
+ service.forHelper(root, { role: Role.Analyst })
375
+ expect(resolved.at(-1)!.override?.maxTokens).toBeUndefined()
241
376
  })
242
377
  })
@@ -187,6 +187,9 @@ describe('helpers/spectate', () => {
187
187
 
188
188
  const last = spectator.entries[0]!.messages.at(-1)!
189
189
  expect(last.contentType).toBe('tool_call')
190
- expect(last.content).toEqual([{ id: '1', name: 'extract', args: { a: 1 } }] as unknown as string)
190
+ // Serialized, not the raw array: a trace row is a string column, and handing it an object
191
+ // is what stored `[object Object]` for every tool call until this was fixed.
192
+ expect(typeof last.content).toBe('string')
193
+ expect(JSON.parse(last.content)).toEqual([{ id: '1', name: 'extract', args: { a: 1 } }])
191
194
  })
192
195
  })
@@ -9,8 +9,12 @@ import {
9
9
  registerLlmPlugin, resolvePlugin,
10
10
  } from '@owlmeans/llm'
11
11
  import type { LlmPlugin, ModelConfig } from '@owlmeans/llm'
12
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
13
+ import { DEFAULT_MAX_OUTPUT_CAP } from '@owlmeans/llm'
12
14
  import { offlineConfigs, Role } from './context.js'
13
15
  import { stripCacheMarkers } from '../src/utils/prompt.js'
16
+ import { resolveOutputCap } from '../src/utils/config.js'
17
+ import { ADAPTIVE_MIN_MAX_TOKENS } from '../src/plugins/anthropic.js'
14
18
 
15
19
  const build = (plugin: LlmPlugin, config: Partial<ModelConfig> = {}) =>
16
20
  plugin.build({
@@ -146,6 +150,30 @@ describe('@owlmeans/llm — retry escalation behaviour', () => {
146
150
  expect((openAiPlugin.refine({ base, attempt: 8, maxOutputCap: 5000 }) as ChatOpenAI).maxTokens).toBe(5000)
147
151
  })
148
152
 
153
+ /**
154
+ * `refine` rebuilds the model for EVERY attempt, attempt 0 included, so a parameter the
155
+ * Responses API rejects has to be suppressed in both hooks: omitting it in `build` alone
156
+ * still put `temperature` on every single request and 400'd the whole gpt-5 family.
157
+ */
158
+ test('refine never restores sampling knobs on a Responses-API model', () => {
159
+ const base = build(openAiPlugin, { model: 'gpt-5.6-terra', maxTokens: 1000 })
160
+ const read = (model: unknown) => model as ChatOpenAI & { topP?: number }
161
+
162
+ for (const attempt of [0, 1, 4]) {
163
+ const refined = read(openAiPlugin.refine({ base, attempt, temperature: 0.7, maxOutputCap: 8000 }))
164
+ expect(refined.temperature).toBeUndefined()
165
+ expect(refined.topP).toBeUndefined()
166
+ expect((refined.lc_kwargs as { useResponsesApi?: boolean }).useResponsesApi).toBe(true)
167
+ }
168
+
169
+ // The chat-completions families still get the deterministic default and the escalator.
170
+ const chat = read(openAiPlugin.refine({
171
+ base: build(openAiPlugin, { model: 'gpt-4.1-mini', maxTokens: 1000 }),
172
+ attempt: 0, maxOutputCap: 8000,
173
+ }))
174
+ expect(chat.temperature).toBe(0)
175
+ })
176
+
149
177
  // Extra budget must become visible output, not more hidden reasoning.
150
178
  test('refine shrinks an absolute reasoning cap as the attempt grows', () => {
151
179
  const base = build(compatiblePlugin, {
@@ -513,3 +541,186 @@ describe('@owlmeans/llm — service', () => {
513
541
  expect(() => service.getModel('no-such-role')).toThrow()
514
542
  })
515
543
  })
544
+
545
+ /**
546
+ * A `preset` is a BASE its referent refines, not a final word. Asserted as a full ladder
547
+ * because the failure it guards was silent: with the preset assigned last, a role that
548
+ * declared one discarded its own fields AND the caller's override, so effort-tier token
549
+ * caps and a temperature refinement simply vanished.
550
+ */
551
+ describe('@owlmeans/llm — config precedence', () => {
552
+ const layered = (): ModelConfig[] => [
553
+ {
554
+ alias: 'base', provider: ModelProvider.OpenAI, model: 'base-model', secret: 'sk-test',
555
+ maxTokens: 1000, temperature: 0.3, topP: 0.5,
556
+ },
557
+ { alias: 'role', preset: 'base', provider: ModelProvider.OpenAI, secret: 'sk-test', maxTokens: 2000 },
558
+ {
559
+ alias: 'other', provider: ModelProvider.OpenAI, model: 'other-model', secret: 'sk-test',
560
+ maxTokens: 500,
561
+ },
562
+ ]
563
+ const configOf = (model: BaseChatModel) =>
564
+ (model as unknown as { metadata: { config: ModelConfig } }).metadata.config
565
+
566
+ test('an alias refines its preset instead of being overwritten by it', () => {
567
+ const service = makeLlmService({ models: layered }, 'spec-prec-alias')
568
+ const config = configOf(service.getModel('role'))
569
+
570
+ expect(config.model).toBe('base-model') // inherited
571
+ expect(config.maxTokens).toBe(2000) // the alias's own field wins
572
+ expect(config.temperature).toBe(0.3) // inherited
573
+ })
574
+
575
+ test('a call override outranks both the alias and its preset', () => {
576
+ const service = makeLlmService({ models: layered }, 'spec-prec-override')
577
+ const config = configOf(service.getModel('role', { maxTokens: 3000, temperature: 0.1 }))
578
+
579
+ expect(config.maxTokens).toBe(3000)
580
+ expect(config.temperature).toBe(0.1)
581
+ expect(config.model).toBe('base-model')
582
+ })
583
+
584
+ test('an override naming a preset picks that model but yields to explicit fields', () => {
585
+ const service = makeLlmService({ models: layered }, 'spec-prec-pin')
586
+ const config = configOf(service.getModel('other', { preset: 'base', maxTokensCap: 32000 }))
587
+
588
+ expect(config.model).toBe('base-model') // the pin outranks the alias's own model
589
+ expect(config.maxTokensCap).toBe(32000) // explicit override field survives the pin
590
+ expect(config.alias).toBe('other') // the alias asked for is what is built
591
+ })
592
+
593
+ test('preset resolution stays one level deep', () => {
594
+ // `role` inherits its model FROM `base`, so pinning `role` contributes only the
595
+ // fields `role` itself declares. One level is the long-standing rule; the layering
596
+ // fix did not make it a chain, and a preset meant to carry a model must name one.
597
+ const service = makeLlmService({ models: layered }, 'spec-prec-depth')
598
+ const config = configOf(service.getModel('other', { preset: 'role' }))
599
+
600
+ expect(config.model).toBe('other-model')
601
+ expect(config.maxTokens).toBe(2000)
602
+ })
603
+
604
+ test('an undefined override value does not shadow the layer below', () => {
605
+ const service = makeLlmService({ models: layered }, 'spec-prec-undef')
606
+ const config = configOf(service.getModel('role', { maxTokens: undefined }))
607
+
608
+ expect(config.maxTokens).toBe(2000)
609
+ })
610
+
611
+ test('the service-wide stream timeout stays a floor under the merge', () => {
612
+ const service = makeLlmService(
613
+ { models: layered, streamTimeout: 12_000 }, 'spec-prec-timeout'
614
+ )
615
+ expect(configOf(service.getModel('role')).streamTimeout).toBe(12_000)
616
+ })
617
+ })
618
+
619
+ describe('@owlmeans/llm — output capability', () => {
620
+ const capped = (): ModelConfig[] => [
621
+ {
622
+ alias: 'small', provider: ModelProvider.OpenAI, model: 'small-model', secret: 'sk-test',
623
+ maxTokens: 16000, maxTokensCap: 64000, maxOutput: 8000, contextWindow: 200_000,
624
+ },
625
+ {
626
+ alias: 'honest', provider: ModelProvider.OpenAI, model: 'honest-model', secret: 'sk-test',
627
+ maxTokens: 4000, maxTokensCap: 32000, maxOutput: 64000, contextWindow: 200_000,
628
+ fallback: { model: 'big-model', maxOutput: 128_000, contextWindow: 1_000_000 },
629
+ },
630
+ ]
631
+
632
+ test('the declared cap chooses the ceiling and the capability trims it', () => {
633
+ expect(resolveOutputCap({ maxTokensCap: 32000 })).toBe(32000)
634
+ expect(resolveOutputCap({ maxOutput: 64000 })).toBe(64000)
635
+ expect(resolveOutputCap({ maxTokensCap: 64000, maxOutput: 8000 })).toBe(8000)
636
+ expect(resolveOutputCap({ maxTokensCap: 16000, maxOutput: 64000 })).toBe(16000)
637
+ expect(resolveOutputCap({})).toBe(DEFAULT_MAX_OUTPUT_CAP)
638
+ })
639
+
640
+ test('an initial budget above the provider capability is clamped at build time', () => {
641
+ const service = makeLlmService({ models: capped }, 'spec-cap-clamp')
642
+ const config = (service.getModel('small') as unknown as { metadata: { config: ModelConfig } })
643
+ .metadata.config
644
+
645
+ expect(config.maxTokens).toBe(8000)
646
+ })
647
+
648
+ test('a fallback carries its own capability rather than the primary\'s', () => {
649
+ const service = makeLlmService({ models: capped }, 'spec-cap-fallback')
650
+ const primary = service.getModel('honest')
651
+ const fallback = (primary as unknown as { __fallbackModel?: BaseChatModel }).__fallbackModel!
652
+ const config = (fallback as unknown as { metadata: { config: ModelConfig } }).metadata.config
653
+
654
+ expect(config.maxOutput).toBe(128_000)
655
+ expect(resolveOutputCap(config)).toBe(32000)
656
+ })
657
+ })
658
+
659
+ /**
660
+ * The models that took the sampling knobs away also think adaptively whether asked to or not,
661
+ * and that thinking is billed against the same `max_tokens` as the answer. A budget sized for
662
+ * the answer alone gets spent on reasoning, and the completion comes back with no text at all.
663
+ */
664
+ describe('@owlmeans/llm — adaptive-thinking output budget', () => {
665
+ const maxTokensOf = (model: BaseChatModel): number | undefined =>
666
+ (model as unknown as { maxTokens?: number }).maxTokens
667
+
668
+ test('an always-reasoning model gets room for the reasoning AND the answer', () => {
669
+ const model = build(anthropicPlugin, { model: 'claude-sonnet-5', maxTokens: 8192 })
670
+
671
+ expect(maxTokensOf(model)).toBe(ADAPTIVE_MIN_MAX_TOKENS)
672
+ })
673
+
674
+ test('the floor never lowers a preset that asked for more', () => {
675
+ const model = build(anthropicPlugin, {
676
+ model: 'claude-sonnet-5', maxTokens: 64_000, maxTokensCap: 64_000, maxOutput: 128_000,
677
+ })
678
+
679
+ expect(maxTokensOf(model)).toBe(64_000)
680
+ })
681
+
682
+ /** Raising the floor past what the provider accepts turns a retryable empty into a 400. */
683
+ test('the floor is clamped to what the provider accepts', () => {
684
+ const model = build(anthropicPlugin, {
685
+ model: 'claude-sonnet-5', maxTokens: 4096, maxOutput: 8192,
686
+ })
687
+
688
+ expect(maxTokensOf(model)).toBe(8192)
689
+ })
690
+
691
+ test('models that still accept sampling keep the budget their preset asked for', () => {
692
+ const model = build(anthropicPlugin, { model: 'claude-haiku-4-5', maxTokens: 8192 })
693
+
694
+ expect(maxTokensOf(model)).toBe(8192)
695
+ })
696
+ })
697
+
698
+ describe('@owlmeans/llm — anthropic thinking control', () => {
699
+ // `thinkingExplicitlySet` is what decides whether langchain puts `thinking` on the wire at
700
+ // all; the field itself defaults to `disabled` and so proves nothing on its own.
701
+ const wire = (model: unknown) => model as { thinkingExplicitlySet: boolean; thinking: { type: string } }
702
+
703
+ test('disableThinking sends an explicit thinking:disabled to the adaptive family', () => {
704
+ const on = wire(build(anthropicPlugin, { model: 'claude-sonnet-5', disableThinking: true }))
705
+ expect(on.thinkingExplicitlySet).toBe(true)
706
+ expect(on.thinking).toEqual({ type: 'disabled' })
707
+ expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-sonnet-5', disableThinking: true } as ModelConfig)).toBe(true)
708
+ })
709
+
710
+ test('without the flag nothing is sent and the provider default applies', () => {
711
+ expect(wire(build(anthropicPlugin, { model: 'claude-sonnet-5' })).thinkingExplicitlySet).toBe(false)
712
+ expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-sonnet-5' } as ModelConfig)).toBe(false)
713
+ })
714
+
715
+ test('models that reason only when asked are left alone', () => {
716
+ expect(wire(build(anthropicPlugin, { model: 'claude-haiku-4-5', disableThinking: true })).thinkingExplicitlySet).toBe(false)
717
+ expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-haiku-4-5', disableThinking: true } as ModelConfig)).toBe(false)
718
+ })
719
+
720
+ test('refine keeps thinking off on every attempt', () => {
721
+ const base = build(anthropicPlugin, { model: 'claude-sonnet-5', disableThinking: true, maxTokens: 1000 })
722
+ const refined = wire(anthropicPlugin.refine({ base, attempt: 1, maxOutputCap: 64000 }))
723
+ expect(refined.thinkingExplicitlySet).toBe(true)
724
+ expect(refined.thinking).toEqual({ type: 'disabled' })
725
+ })
726
+ })
@@ -148,6 +148,49 @@ describe('@owlmeans/llm — prompt composition', () => {
148
148
  expect(context[0]!.text).toContain('second note')
149
149
  })
150
150
 
151
+ // Two plugins can each be able to render the same skill — a static catalogue and a
152
+ // detector. Without the claim they both do, and the model is told the same thing twice.
153
+ test('a key is claimed by the first plugin only, within one composition', async () => {
154
+ const grants: boolean[] = []
155
+ const claimer = (alias: string): LlmPromptPlugin => ({
156
+ alias,
157
+ compose: ctx => {
158
+ if (ctx.claim('skill:a')) {
159
+ ctx.add(PromptBlock.Packages, `from ${alias}`)
160
+ }
161
+ grants.push(ctx.claim('probe'))
162
+ },
163
+ })
164
+ const result = await compose(service([], [claimer('first'), claimer('second')]))
165
+
166
+ expect(grants).toEqual([true, false])
167
+ expect(result.blocks.find(b => b.block === PromptBlock.Packages)?.text).toBe('from first')
168
+ })
169
+
170
+ test('the claim set is per composition, not per service', async () => {
171
+ const svc = service([], [{
172
+ alias: 'claimer', compose: ctx => ctx.add(PromptBlock.Packages, `${ctx.claim('once')}`),
173
+ }])
174
+ const first = await compose(svc)
175
+ const second = await compose(svc)
176
+
177
+ expect(first.blocks[0]!.text).toBe('true')
178
+ expect(second.blocks[0]!.text).toBe('true')
179
+ })
180
+
181
+ // The seam has to be free: a composition where nobody claims must render the bytes it
182
+ // rendered before the seam existed, or every cached prefix in the fleet is invalidated.
183
+ test('claiming nothing changes nothing about the composed bytes', async () => {
184
+ const input = { role: 'R', skills: ['a'], context: ['note'] }
185
+ const plain = await compose(service([skill('a', 'A')]), input)
186
+ const claiming = await compose(
187
+ service([skill('a', 'A')], [{ alias: 'quiet', compose: ctx => { ctx.claim('a') } }]),
188
+ input,
189
+ )
190
+ expect(JSON.stringify(claiming.system)).toBe(JSON.stringify(plain.system))
191
+ expect(claiming.breakpoints).toBe(plain.breakpoints)
192
+ })
193
+
151
194
  test('nothing declared composes to no system message at all', async () => {
152
195
  const result = await compose(service())
153
196
  expect(result.system).toBeNull()