@owlmeans/llm 0.1.18-rc.2 → 0.1.18-rc.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/agent-meta/manifest.json +2 -2
- package/agent-meta/skills/llm/SKILL.md +239 -116
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +37 -3
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +46 -11
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +72 -12
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +15 -9
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/retry.d.ts +14 -0
- package/build/helpers/retry.d.ts.map +1 -1
- package/build/helpers/retry.js +14 -0
- package/build/helpers/retry.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +12 -5
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +2 -2
- package/build/index.d.ts.map +1 -1
- package/build/index.js +2 -2
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +60 -28
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts +40 -0
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +79 -6
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +12 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +29 -3
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +7 -0
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/service.d.ts.map +1 -1
- package/build/prompt/service.js +11 -0
- package/build/prompt/service.js.map +1 -1
- package/build/prompt/types.d.ts +20 -0
- package/build/prompt/types.d.ts.map +1 -1
- package/build/service.d.ts.map +1 -1
- package/build/service.js +42 -7
- package/build/service.js.map +1 -1
- package/build/types.d.ts +71 -1
- package/build/types.d.ts.map +1 -1
- package/build/utils/config.d.ts +11 -0
- package/build/utils/config.d.ts.map +1 -1
- package/build/utils/config.js +21 -1
- package/build/utils/config.js.map +1 -1
- package/build/utils/null-report.d.ts.map +1 -1
- package/build/utils/null-report.js +9 -1
- package/build/utils/null-report.js.map +1 -1
- package/package.json +13 -13
- package/src/execution/service.ts +46 -11
- package/src/execution/types.ts +77 -13
- package/src/execution/utils.ts +16 -9
- package/src/helpers/retry.ts +16 -0
- package/src/helpers/spectate.ts +14 -5
- package/src/index.ts +4 -2
- package/src/model.ts +77 -27
- package/src/plugins/anthropic.ts +89 -6
- package/src/plugins/openai.ts +33 -3
- package/src/plugins/types.ts +8 -0
- package/src/prompt/service.ts +12 -0
- package/src/prompt/types.ts +20 -0
- package/src/service.ts +53 -7
- package/src/types.ts +71 -1
- package/src/utils/config.ts +23 -1
- package/src/utils/null-report.ts +9 -1
- package/tests/context.ts +3 -0
- package/tests/execution.spec.ts +148 -13
- package/tests/helpers.spec.ts +4 -1
- package/tests/plugins.spec.ts +211 -0
- package/tests/prompt.spec.ts +43 -0
package/src/types.ts
CHANGED
|
@@ -63,6 +63,12 @@ export interface LlmModelOptions extends LlmLogging {
|
|
|
63
63
|
prompts?: () => PromptService
|
|
64
64
|
/** File access offered to prompt plugins that resolve knowledge from disk. */
|
|
65
65
|
files?: FileProviderRef
|
|
66
|
+
/**
|
|
67
|
+
* Cheap model offered to prompt plugins for a single side call while composing —
|
|
68
|
+
* normally `() => executions().utility(exec)`. Omitted, plugins that would use one
|
|
69
|
+
* fall back to whatever they can decide without a model.
|
|
70
|
+
*/
|
|
71
|
+
utility?: () => BaseChatModel | undefined
|
|
66
72
|
}
|
|
67
73
|
|
|
68
74
|
export type ModelMessage = BaseMessage | MessageFieldWithRole
|
|
@@ -89,6 +95,27 @@ export interface LlmCallOptions {
|
|
|
89
95
|
* should be part of the shared prefix.
|
|
90
96
|
*/
|
|
91
97
|
skills?: string[]
|
|
98
|
+
/**
|
|
99
|
+
* How many failures of this SAME piece of work the caller has already observed — the
|
|
100
|
+
* per-call escalator starts that far up its ladder instead of at the bottom.
|
|
101
|
+
*
|
|
102
|
+
* A caller that validates the OUTPUT (a diff that must apply, a file that must not come
|
|
103
|
+
* back truncated) runs its own retry loop around whole calls, and every one of those
|
|
104
|
+
* calls used to start at attempt 0: same model, same output budget, same answer. Passing
|
|
105
|
+
* the outer attempt here advances both rungs the escalator owns — the `maxTokens`
|
|
106
|
+
* doubling and the {@link FALLBACK_AFTER_ATTEMPTS} switch to `ModelConfig.fallback` — so
|
|
107
|
+
* a repeatedly-rejected call actually reaches the stronger model.
|
|
108
|
+
*
|
|
109
|
+
* Clamped to `retries - 1`; it moves the STARTING rung only and never changes how many
|
|
110
|
+
* attempts this call makes. Unrelated to `ExecutionService.escalate`, which raises the
|
|
111
|
+
* effort tier of an execution before any model is resolved.
|
|
112
|
+
*/
|
|
113
|
+
escalation?: number
|
|
114
|
+
/**
|
|
115
|
+
* Abort THIS call's retry loop for an error no retry can fix. Consulted before the
|
|
116
|
+
* globally registered resolvers and the provider plugin's `isFatal`.
|
|
117
|
+
*/
|
|
118
|
+
fatal?: (e: unknown) => Error | null
|
|
92
119
|
}
|
|
93
120
|
|
|
94
121
|
export interface LlmAskOptions extends LlmCallOptions {
|
|
@@ -151,20 +178,63 @@ export interface TemperatureFactory {
|
|
|
151
178
|
export interface ModelConfig {
|
|
152
179
|
provider?: ModelProvider | string
|
|
153
180
|
secret?: string
|
|
181
|
+
/**
|
|
182
|
+
* Which delegate transport answers this model's calls.
|
|
183
|
+
*
|
|
184
|
+
* Only meaningful for {@link ModelProvider.Delegated}: the key the application seated a
|
|
185
|
+
* transport under, so one deployment can hold many at once — one per connected agent — and a
|
|
186
|
+
* config names the one that belongs to its run.
|
|
187
|
+
*/
|
|
188
|
+
delegate?: string
|
|
154
189
|
alias: string
|
|
155
190
|
/** Inherit every field of another alias in the same config list. */
|
|
156
191
|
preset?: string
|
|
157
192
|
model?: string
|
|
158
193
|
temperature?: number
|
|
159
|
-
/**
|
|
194
|
+
/**
|
|
195
|
+
* Initial output-token budget per request (`max_tokens`) — what THIS deployment asks
|
|
196
|
+
* for first, not what the model can do. Clamped to {@link ModelConfig.maxOutput}.
|
|
197
|
+
*/
|
|
160
198
|
maxTokens?: number
|
|
161
199
|
/**
|
|
162
200
|
* Hard ceiling for output tokens used by the retry escalator. Each retry doubles
|
|
163
201
|
* `maxTokens` toward this cap; without it the escalator uses
|
|
164
202
|
* {@link DEFAULT_MAX_OUTPUT_CAP}, which exceeds many models' real per-request output
|
|
165
203
|
* limit and turns retries into 400 "max_tokens exceeds model limit" errors.
|
|
204
|
+
*
|
|
205
|
+
* This is the budget the PRESET chooses; {@link ModelConfig.maxOutput} is what the
|
|
206
|
+
* provider allows. The effective ceiling is the smaller of the two, so a preset that
|
|
207
|
+
* over-declares is corrected at refine time rather than at the provider.
|
|
166
208
|
*/
|
|
167
209
|
maxTokensCap?: number
|
|
210
|
+
/**
|
|
211
|
+
* Total context window the model accepts — input plus output. Informational: it is
|
|
212
|
+
* never sent to the provider and the runtime cannot enforce it (the input size is not
|
|
213
|
+
* known at config time). It documents the model and lets the preset test catch a
|
|
214
|
+
* `maxOutput` that could not possibly fit.
|
|
215
|
+
*/
|
|
216
|
+
contextWindow?: number
|
|
217
|
+
/**
|
|
218
|
+
* What the PROVIDER accepts as output in a single request. Unlike `maxTokens` and
|
|
219
|
+
* `maxTokensCap` — both deployment choices — this is a property of the model behind
|
|
220
|
+
* this alias, and for an aggregated model it is the limit of the `inferenceProvider`
|
|
221
|
+
* actually pinned here, which is often well below what the model can do elsewhere.
|
|
222
|
+
*
|
|
223
|
+
* Hard ceiling: `maxTokens` is clamped to it at build time and the escalator's cap is
|
|
224
|
+
* `min(maxTokensCap, maxOutput)`.
|
|
225
|
+
*
|
|
226
|
+
* A `fallback` that changes `model` MUST redeclare this (and `contextWindow`), because
|
|
227
|
+
* the fallback config inherits every field the patch does not name — otherwise the
|
|
228
|
+
* escalator would size the fallback by the primary's capability.
|
|
229
|
+
*/
|
|
230
|
+
maxOutput?: number
|
|
231
|
+
/**
|
|
232
|
+
* The context window is SHARED between input and output rather than being an input
|
|
233
|
+
* allowance with a separate output limit (MiniMax M2.x, gpt-oss). Nothing enforces it
|
|
234
|
+
* at runtime; it marks the entry so `maxOutput` is read as "spends the same budget the
|
|
235
|
+
* prompt spends" and keeps presets honest about leaving room for input.
|
|
236
|
+
*/
|
|
237
|
+
combinedWindow?: boolean
|
|
168
238
|
topP?: number
|
|
169
239
|
baseUrl?: string
|
|
170
240
|
organization?: string
|
package/src/utils/config.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
|
-
import { MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
|
|
2
|
+
import { DEFAULT_MAX_OUTPUT_CAP, MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
|
|
3
3
|
import type { ModelConfig } from '../types.js'
|
|
4
4
|
|
|
5
5
|
/**
|
|
@@ -17,3 +17,25 @@ export const readConfig = (model: BaseChatModel): Partial<ModelConfig> => {
|
|
|
17
17
|
/** Per-model idle stream timeout (ms), falling back to the package default. */
|
|
18
18
|
export const idleTimeout = (config: Partial<ModelConfig>): number =>
|
|
19
19
|
config.streamTimeout ?? MODEL_STREAM_TIMEOUT_MS
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* The output ceiling the retry escalator may climb to.
|
|
23
|
+
*
|
|
24
|
+
* Two different things claim to bound the output, and only one of them is a fact:
|
|
25
|
+
* `maxTokensCap` is what this deployment budgeted, `maxOutput` is what the provider will
|
|
26
|
+
* accept. Taking the declared cap alone let a preset out-declare its model and turned
|
|
27
|
+
* every escalation into a 400; taking the capability alone would ignore a deliberately
|
|
28
|
+
* frugal budget. So the declared value chooses the ceiling and the capability trims it,
|
|
29
|
+
* and only when neither is stated does {@link DEFAULT_MAX_OUTPUT_CAP} apply.
|
|
30
|
+
*/
|
|
31
|
+
export const resolveOutputCap = (config: Partial<ModelConfig>): number => {
|
|
32
|
+
const declared = typeof config.maxTokensCap === 'number' && config.maxTokensCap > 0
|
|
33
|
+
? config.maxTokensCap
|
|
34
|
+
: undefined
|
|
35
|
+
const capability = typeof config.maxOutput === 'number' && config.maxOutput > 0
|
|
36
|
+
? config.maxOutput
|
|
37
|
+
: undefined
|
|
38
|
+
const cap = declared ?? capability ?? DEFAULT_MAX_OUTPUT_CAP
|
|
39
|
+
|
|
40
|
+
return capability != null ? Math.min(cap, capability) : cap
|
|
41
|
+
}
|
package/src/utils/null-report.ts
CHANGED
|
@@ -45,6 +45,7 @@ export const buildNullReport = (p: NullReportParams): NullCapture => {
|
|
|
45
45
|
finish_reason?: string
|
|
46
46
|
usage?: { prompt_tokens?: number; completion_tokens?: number; reasoning_tokens?: number }
|
|
47
47
|
} | undefined
|
|
48
|
+
const stopReason = (raw?.additional_kwargs as { stop_reason?: string } | undefined)?.stop_reason
|
|
48
49
|
const usageMeta = raw?.usage_metadata as { input_tokens?: number; output_tokens?: number } | undefined
|
|
49
50
|
const toolCalls = (raw as unknown as { tool_calls?: unknown[] } | null)?.tool_calls
|
|
50
51
|
|
|
@@ -81,7 +82,14 @@ export const buildNullReport = (p: NullReportParams): NullCapture => {
|
|
|
81
82
|
tool_calls: toolCalls,
|
|
82
83
|
} : null,
|
|
83
84
|
diagnostics: {
|
|
84
|
-
|
|
85
|
+
// Two providers, two places. OpenAI-compatible APIs put it on `response_metadata`;
|
|
86
|
+
// Anthropic never does — it arrives on the `message_delta` event and langchain spreads it
|
|
87
|
+
// into `additional_kwargs.stop_reason`. Reading only the first printed `undefined` for
|
|
88
|
+
// every Anthropic null, hiding the `max_tokens` that explains most of them.
|
|
89
|
+
finishReason: responseMeta?.finish_reason ?? stopReason,
|
|
90
|
+
/** No text block at all — the shape of a completion that was all reasoning. */
|
|
91
|
+
thinkingOnly: Array.isArray(raw?.content) && raw.content.length > 0
|
|
92
|
+
&& !raw.content.some(part => (part as { type?: string }).type === 'text'),
|
|
85
93
|
inputTokens: usageMeta?.input_tokens ?? responseMeta?.usage?.prompt_tokens,
|
|
86
94
|
outputTokens: usageMeta?.output_tokens ?? responseMeta?.usage?.completion_tokens,
|
|
87
95
|
reasoningTokens: responseMeta?.usage?.reasoning_tokens,
|
package/tests/context.ts
CHANGED
|
@@ -16,6 +16,8 @@ export const gates = makeGates({
|
|
|
16
16
|
export const Role = {
|
|
17
17
|
Analyst: 'analyst',
|
|
18
18
|
Picker: 'picker',
|
|
19
|
+
/** The conventional cheap tier — same value as `UTILITY_ROLE`. */
|
|
20
|
+
Utility: 'utility',
|
|
19
21
|
} as const
|
|
20
22
|
|
|
21
23
|
/**
|
|
@@ -60,6 +62,7 @@ export const anthropicConfigs = (): ModelConfig[] => [
|
|
|
60
62
|
*/
|
|
61
63
|
export const offlineConfigs = (): ModelConfig[] => [
|
|
62
64
|
{ alias: Role.Analyst, provider: ModelProvider.OpenAI, model: 'gpt-4.1-mini', secret: 'sk-test' },
|
|
65
|
+
{ alias: Role.Utility, provider: ModelProvider.OpenAI, model: 'gpt-4.1-nano', secret: 'sk-test' },
|
|
63
66
|
{
|
|
64
67
|
alias: Role.Picker, provider: ModelProvider.Compatible, model: 'some/model', secret: 'sk-test',
|
|
65
68
|
baseUrl: 'https://openrouter.ai/api/v1',
|
package/tests/execution.spec.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { beforeEach, describe, expect, test } from 'bun:test'
|
|
2
|
-
import { ExecutionEffort, ExecutionLevel } from '@owlmeans/llm-common'
|
|
2
|
+
import { ExecutionEffort, ExecutionLevel, UTILITY_ROLE } from '@owlmeans/llm-common'
|
|
3
3
|
import type { ExecutionState, ModelConfigPatch, TaskExecutionState } from '@owlmeans/llm-common'
|
|
4
4
|
import { DEFAULT_EFFORT, EFFORT_TABLE, makeExecutionService, makeLlmService } from '@owlmeans/llm'
|
|
5
5
|
import type { ExecutionService, ProjectExecution, TaskExecution } from '@owlmeans/llm'
|
|
@@ -135,6 +135,65 @@ describe('@owlmeans/llm — model policy resolution', () => {
|
|
|
135
135
|
})
|
|
136
136
|
})
|
|
137
137
|
|
|
138
|
+
describe('@owlmeans/llm — utility tier', () => {
|
|
139
|
+
test('the conventional role is resolved at the cheapest tier', () => {
|
|
140
|
+
service.utility(root)
|
|
141
|
+
expect(resolved.at(-1))
|
|
142
|
+
.toEqual({ alias: UTILITY_ROLE, override: EFFORT_TABLE[ExecutionEffort.Economy] })
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
// The whole point of routing it through `model`: a deployment remaps or pins the cheap
|
|
146
|
+
// tier with the levers it already uses for every other role.
|
|
147
|
+
test('a role override remaps which model the utility tier resolves', () => {
|
|
148
|
+
const remapped = service.escalate(root, { roleOverrides: { [UTILITY_ROLE]: Role.Picker } })
|
|
149
|
+
service.utility(remapped)
|
|
150
|
+
expect(resolved.at(-1)?.alias).toBe(Role.Picker)
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
test('a policy utility role replaces the conventional one, and is still remapped', () => {
|
|
154
|
+
const named = service.root({
|
|
155
|
+
models: makeService(),
|
|
156
|
+
policy: { effort: DEFAULT_EFFORT, utilityRole: Role.Analyst },
|
|
157
|
+
purpose: { type: 'spec' },
|
|
158
|
+
})
|
|
159
|
+
service.utility(named)
|
|
160
|
+
expect(resolved.at(-1)?.alias).toBe(Role.Analyst)
|
|
161
|
+
|
|
162
|
+
service.utility(service.escalate(named, { roleOverrides: { [Role.Analyst]: Role.Picker } }))
|
|
163
|
+
expect(resolved.at(-1)?.alias).toBe(Role.Picker)
|
|
164
|
+
})
|
|
165
|
+
|
|
166
|
+
test('a model override pins the utility role like any other', () => {
|
|
167
|
+
const pinned = service.escalate(root, { modelOverrides: { [UTILITY_ROLE]: { maxTokens: 512 } } })
|
|
168
|
+
service.utility(pinned)
|
|
169
|
+
expect(resolved.at(-1)?.override)
|
|
170
|
+
.toEqual({ ...EFFORT_TABLE[ExecutionEffort.Economy], maxTokens: 512 })
|
|
171
|
+
})
|
|
172
|
+
|
|
173
|
+
// The floor is local — asking for a cheap side model must not quietly downgrade the
|
|
174
|
+
// execution that the real work still runs on.
|
|
175
|
+
test('the economy floor does not touch the execution it was asked on', () => {
|
|
176
|
+
const high = service.escalate(root, { effort: ExecutionEffort.Max })
|
|
177
|
+
service.utility(high)
|
|
178
|
+
expect(high.policy.effort).toBe(ExecutionEffort.Max)
|
|
179
|
+
|
|
180
|
+
service.model(high, Role.Analyst)
|
|
181
|
+
expect(resolved.at(-1)?.override).toEqual(EFFORT_TABLE[ExecutionEffort.Max])
|
|
182
|
+
})
|
|
183
|
+
|
|
184
|
+
test('a named utility role survives refinement of the branch', () => {
|
|
185
|
+
const named = service.root({
|
|
186
|
+
models: makeService(),
|
|
187
|
+
policy: { effort: DEFAULT_EFFORT, utilityRole: Role.Analyst },
|
|
188
|
+
purpose: { type: 'spec' },
|
|
189
|
+
})
|
|
190
|
+
const task = service.forTask(named, { effort: ExecutionEffort.High })
|
|
191
|
+
expect(task.policy.utilityRole).toBe(Role.Analyst)
|
|
192
|
+
service.utility(task)
|
|
193
|
+
expect(resolved.at(-1)?.alias).toBe(Role.Analyst)
|
|
194
|
+
})
|
|
195
|
+
})
|
|
196
|
+
|
|
138
197
|
describe('@owlmeans/llm — snapshot and restore', () => {
|
|
139
198
|
test('a snapshot carries the state and none of the collaborators', () => {
|
|
140
199
|
const state = service.snapshot(root) as ExecutionState & Record<string, unknown>
|
|
@@ -222,21 +281,97 @@ describe('@owlmeans/llm — prompt policy accumulation', () => {
|
|
|
222
281
|
})
|
|
223
282
|
})
|
|
224
283
|
|
|
225
|
-
describe('@owlmeans/llm —
|
|
226
|
-
test('
|
|
227
|
-
|
|
284
|
+
describe('@owlmeans/llm — plugin registration', () => {
|
|
285
|
+
test('seats a plugin by alias, so a layer wired twice answers once', async () => {
|
|
286
|
+
// Mixins compose. A plugin registered twice would answer twice, and since the first usable
|
|
287
|
+
// answer wins, the duplicate is silent rather than loud.
|
|
288
|
+
const asked: string[] = []
|
|
289
|
+
service.use({ alias: 'files', advise: async () => { asked.push('first'); return null } })
|
|
290
|
+
service.use({ alias: 'files', advise: async () => { asked.push('second'); return 'answer' } })
|
|
291
|
+
|
|
292
|
+
await service.advise(root, { kind: 'files', task: 'x' })
|
|
293
|
+
|
|
294
|
+
expect(asked).toEqual(['second'])
|
|
228
295
|
})
|
|
229
296
|
|
|
230
|
-
test('a
|
|
231
|
-
const
|
|
232
|
-
service.use({
|
|
297
|
+
test('a plugin with no alias is simply appended', async () => {
|
|
298
|
+
const asked: string[] = []
|
|
299
|
+
service.use({ advise: async () => { asked.push('a'); return null } })
|
|
300
|
+
service.use({ advise: async () => { asked.push('b'); return null } })
|
|
233
301
|
|
|
234
|
-
|
|
235
|
-
|
|
302
|
+
await service.advise(root, { kind: 'files', task: 'x' })
|
|
303
|
+
|
|
304
|
+
expect(asked).toEqual(['a', 'b'])
|
|
305
|
+
})
|
|
306
|
+
})
|
|
307
|
+
|
|
308
|
+
describe('@owlmeans/llm — advice seam', () => {
|
|
309
|
+
test('advise resolves to null when no plugin is registered', async () => {
|
|
310
|
+
await expect(service.advise(root, { kind: 'files', task: 'x' })).resolves.toBeNull()
|
|
311
|
+
})
|
|
312
|
+
|
|
313
|
+
test('an advisor receives the execution and the request', async () => {
|
|
314
|
+
const seen: Array<{ kind: string, task: string, level: string }> = []
|
|
315
|
+
service.use({
|
|
316
|
+
advise: async (exec, request) => {
|
|
317
|
+
seen.push({ kind: request.kind, task: request.task, level: exec.level })
|
|
318
|
+
return 'the card'
|
|
319
|
+
},
|
|
320
|
+
})
|
|
321
|
+
|
|
322
|
+
const helper = service.forHelper(root, { role: Role.Analyst })
|
|
323
|
+
await expect(service.advise(helper, { kind: 'files', task: 'implement X' })).resolves.toBe('the card')
|
|
324
|
+
expect(seen).toEqual([{ kind: 'files', task: 'implement X', level: ExecutionLevel.Helper }])
|
|
325
|
+
})
|
|
326
|
+
|
|
327
|
+
test('the first usable answer wins and later advisors are not consulted', async () => {
|
|
328
|
+
let secondCalled = false
|
|
329
|
+
service.use({ advise: async () => null })
|
|
330
|
+
service.use({ advise: async () => ' ' })
|
|
331
|
+
service.use({ advise: async () => 'first real' })
|
|
332
|
+
service.use({ advise: async () => { secondCalled = true; return 'second' } })
|
|
333
|
+
|
|
334
|
+
await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBe('first real')
|
|
335
|
+
expect(secondCalled).toBe(false)
|
|
336
|
+
})
|
|
337
|
+
|
|
338
|
+
test('a throwing advisor is skipped rather than failing the work', async () => {
|
|
339
|
+
service.use({ advise: async () => { throw new Error('advisor down') } })
|
|
340
|
+
service.use({ advise: async () => 'survived' })
|
|
341
|
+
|
|
342
|
+
await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBe('survived')
|
|
343
|
+
})
|
|
344
|
+
|
|
345
|
+
test('a plugin without advise is ignored', async () => {
|
|
346
|
+
service.use({ onCheckpoint: async () => {} })
|
|
347
|
+
await expect(service.advise(root, { kind: 'k', task: 't' })).resolves.toBeNull()
|
|
348
|
+
})
|
|
349
|
+
})
|
|
350
|
+
|
|
351
|
+
describe('@owlmeans/llm — per-helper output sizing', () => {
|
|
352
|
+
test('output becomes the resolved model\'s initial maxTokens', () => {
|
|
353
|
+
service.forHelper(root, { role: Role.Analyst, output: 24000 })
|
|
354
|
+
|
|
355
|
+
const call = resolved.at(-1)!
|
|
356
|
+
expect(call.override?.maxTokens).toBe(24000)
|
|
357
|
+
})
|
|
358
|
+
|
|
359
|
+
test('output is not carried onto the helper as a field', () => {
|
|
360
|
+
const helper = service.forHelper(root, { role: Role.Analyst, output: 24000 })
|
|
361
|
+
expect((helper as unknown as { output?: unknown }).output).toBeUndefined()
|
|
362
|
+
})
|
|
363
|
+
|
|
364
|
+
test('a temperature refinement keeps the helper\'s output budget', () => {
|
|
365
|
+
const helper = service.forHelper(root, { role: Role.Analyst, output: 24000 })
|
|
366
|
+
helper.temperatureFactory(0.7)
|
|
367
|
+
|
|
368
|
+
const call = resolved.at(-1)!
|
|
369
|
+
expect(call.override?.maxTokens).toBe(24000)
|
|
370
|
+
expect(call.override?.temperature).toBe(0.7)
|
|
371
|
+
})
|
|
236
372
|
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
expect((
|
|
240
|
-
expect((seen[0]!.state as unknown as { models?: unknown }).models).toBeUndefined()
|
|
373
|
+
test('without output the override carries no token sizing', () => {
|
|
374
|
+
service.forHelper(root, { role: Role.Analyst })
|
|
375
|
+
expect(resolved.at(-1)!.override?.maxTokens).toBeUndefined()
|
|
241
376
|
})
|
|
242
377
|
})
|
package/tests/helpers.spec.ts
CHANGED
|
@@ -187,6 +187,9 @@ describe('helpers/spectate', () => {
|
|
|
187
187
|
|
|
188
188
|
const last = spectator.entries[0]!.messages.at(-1)!
|
|
189
189
|
expect(last.contentType).toBe('tool_call')
|
|
190
|
-
|
|
190
|
+
// Serialized, not the raw array: a trace row is a string column, and handing it an object
|
|
191
|
+
// is what stored `[object Object]` for every tool call until this was fixed.
|
|
192
|
+
expect(typeof last.content).toBe('string')
|
|
193
|
+
expect(JSON.parse(last.content)).toEqual([{ id: '1', name: 'extract', args: { a: 1 } }])
|
|
191
194
|
})
|
|
192
195
|
})
|
package/tests/plugins.spec.ts
CHANGED
|
@@ -9,8 +9,12 @@ import {
|
|
|
9
9
|
registerLlmPlugin, resolvePlugin,
|
|
10
10
|
} from '@owlmeans/llm'
|
|
11
11
|
import type { LlmPlugin, ModelConfig } from '@owlmeans/llm'
|
|
12
|
+
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
13
|
+
import { DEFAULT_MAX_OUTPUT_CAP } from '@owlmeans/llm'
|
|
12
14
|
import { offlineConfigs, Role } from './context.js'
|
|
13
15
|
import { stripCacheMarkers } from '../src/utils/prompt.js'
|
|
16
|
+
import { resolveOutputCap } from '../src/utils/config.js'
|
|
17
|
+
import { ADAPTIVE_MIN_MAX_TOKENS } from '../src/plugins/anthropic.js'
|
|
14
18
|
|
|
15
19
|
const build = (plugin: LlmPlugin, config: Partial<ModelConfig> = {}) =>
|
|
16
20
|
plugin.build({
|
|
@@ -146,6 +150,30 @@ describe('@owlmeans/llm — retry escalation behaviour', () => {
|
|
|
146
150
|
expect((openAiPlugin.refine({ base, attempt: 8, maxOutputCap: 5000 }) as ChatOpenAI).maxTokens).toBe(5000)
|
|
147
151
|
})
|
|
148
152
|
|
|
153
|
+
/**
|
|
154
|
+
* `refine` rebuilds the model for EVERY attempt, attempt 0 included, so a parameter the
|
|
155
|
+
* Responses API rejects has to be suppressed in both hooks: omitting it in `build` alone
|
|
156
|
+
* still put `temperature` on every single request and 400'd the whole gpt-5 family.
|
|
157
|
+
*/
|
|
158
|
+
test('refine never restores sampling knobs on a Responses-API model', () => {
|
|
159
|
+
const base = build(openAiPlugin, { model: 'gpt-5.6-terra', maxTokens: 1000 })
|
|
160
|
+
const read = (model: unknown) => model as ChatOpenAI & { topP?: number }
|
|
161
|
+
|
|
162
|
+
for (const attempt of [0, 1, 4]) {
|
|
163
|
+
const refined = read(openAiPlugin.refine({ base, attempt, temperature: 0.7, maxOutputCap: 8000 }))
|
|
164
|
+
expect(refined.temperature).toBeUndefined()
|
|
165
|
+
expect(refined.topP).toBeUndefined()
|
|
166
|
+
expect((refined.lc_kwargs as { useResponsesApi?: boolean }).useResponsesApi).toBe(true)
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// The chat-completions families still get the deterministic default and the escalator.
|
|
170
|
+
const chat = read(openAiPlugin.refine({
|
|
171
|
+
base: build(openAiPlugin, { model: 'gpt-4.1-mini', maxTokens: 1000 }),
|
|
172
|
+
attempt: 0, maxOutputCap: 8000,
|
|
173
|
+
}))
|
|
174
|
+
expect(chat.temperature).toBe(0)
|
|
175
|
+
})
|
|
176
|
+
|
|
149
177
|
// Extra budget must become visible output, not more hidden reasoning.
|
|
150
178
|
test('refine shrinks an absolute reasoning cap as the attempt grows', () => {
|
|
151
179
|
const base = build(compatiblePlugin, {
|
|
@@ -513,3 +541,186 @@ describe('@owlmeans/llm — service', () => {
|
|
|
513
541
|
expect(() => service.getModel('no-such-role')).toThrow()
|
|
514
542
|
})
|
|
515
543
|
})
|
|
544
|
+
|
|
545
|
+
/**
|
|
546
|
+
* A `preset` is a BASE its referent refines, not a final word. Asserted as a full ladder
|
|
547
|
+
* because the failure it guards was silent: with the preset assigned last, a role that
|
|
548
|
+
* declared one discarded its own fields AND the caller's override, so effort-tier token
|
|
549
|
+
* caps and a temperature refinement simply vanished.
|
|
550
|
+
*/
|
|
551
|
+
describe('@owlmeans/llm — config precedence', () => {
|
|
552
|
+
const layered = (): ModelConfig[] => [
|
|
553
|
+
{
|
|
554
|
+
alias: 'base', provider: ModelProvider.OpenAI, model: 'base-model', secret: 'sk-test',
|
|
555
|
+
maxTokens: 1000, temperature: 0.3, topP: 0.5,
|
|
556
|
+
},
|
|
557
|
+
{ alias: 'role', preset: 'base', provider: ModelProvider.OpenAI, secret: 'sk-test', maxTokens: 2000 },
|
|
558
|
+
{
|
|
559
|
+
alias: 'other', provider: ModelProvider.OpenAI, model: 'other-model', secret: 'sk-test',
|
|
560
|
+
maxTokens: 500,
|
|
561
|
+
},
|
|
562
|
+
]
|
|
563
|
+
const configOf = (model: BaseChatModel) =>
|
|
564
|
+
(model as unknown as { metadata: { config: ModelConfig } }).metadata.config
|
|
565
|
+
|
|
566
|
+
test('an alias refines its preset instead of being overwritten by it', () => {
|
|
567
|
+
const service = makeLlmService({ models: layered }, 'spec-prec-alias')
|
|
568
|
+
const config = configOf(service.getModel('role'))
|
|
569
|
+
|
|
570
|
+
expect(config.model).toBe('base-model') // inherited
|
|
571
|
+
expect(config.maxTokens).toBe(2000) // the alias's own field wins
|
|
572
|
+
expect(config.temperature).toBe(0.3) // inherited
|
|
573
|
+
})
|
|
574
|
+
|
|
575
|
+
test('a call override outranks both the alias and its preset', () => {
|
|
576
|
+
const service = makeLlmService({ models: layered }, 'spec-prec-override')
|
|
577
|
+
const config = configOf(service.getModel('role', { maxTokens: 3000, temperature: 0.1 }))
|
|
578
|
+
|
|
579
|
+
expect(config.maxTokens).toBe(3000)
|
|
580
|
+
expect(config.temperature).toBe(0.1)
|
|
581
|
+
expect(config.model).toBe('base-model')
|
|
582
|
+
})
|
|
583
|
+
|
|
584
|
+
test('an override naming a preset picks that model but yields to explicit fields', () => {
|
|
585
|
+
const service = makeLlmService({ models: layered }, 'spec-prec-pin')
|
|
586
|
+
const config = configOf(service.getModel('other', { preset: 'base', maxTokensCap: 32000 }))
|
|
587
|
+
|
|
588
|
+
expect(config.model).toBe('base-model') // the pin outranks the alias's own model
|
|
589
|
+
expect(config.maxTokensCap).toBe(32000) // explicit override field survives the pin
|
|
590
|
+
expect(config.alias).toBe('other') // the alias asked for is what is built
|
|
591
|
+
})
|
|
592
|
+
|
|
593
|
+
test('preset resolution stays one level deep', () => {
|
|
594
|
+
// `role` inherits its model FROM `base`, so pinning `role` contributes only the
|
|
595
|
+
// fields `role` itself declares. One level is the long-standing rule; the layering
|
|
596
|
+
// fix did not make it a chain, and a preset meant to carry a model must name one.
|
|
597
|
+
const service = makeLlmService({ models: layered }, 'spec-prec-depth')
|
|
598
|
+
const config = configOf(service.getModel('other', { preset: 'role' }))
|
|
599
|
+
|
|
600
|
+
expect(config.model).toBe('other-model')
|
|
601
|
+
expect(config.maxTokens).toBe(2000)
|
|
602
|
+
})
|
|
603
|
+
|
|
604
|
+
test('an undefined override value does not shadow the layer below', () => {
|
|
605
|
+
const service = makeLlmService({ models: layered }, 'spec-prec-undef')
|
|
606
|
+
const config = configOf(service.getModel('role', { maxTokens: undefined }))
|
|
607
|
+
|
|
608
|
+
expect(config.maxTokens).toBe(2000)
|
|
609
|
+
})
|
|
610
|
+
|
|
611
|
+
test('the service-wide stream timeout stays a floor under the merge', () => {
|
|
612
|
+
const service = makeLlmService(
|
|
613
|
+
{ models: layered, streamTimeout: 12_000 }, 'spec-prec-timeout'
|
|
614
|
+
)
|
|
615
|
+
expect(configOf(service.getModel('role')).streamTimeout).toBe(12_000)
|
|
616
|
+
})
|
|
617
|
+
})
|
|
618
|
+
|
|
619
|
+
describe('@owlmeans/llm — output capability', () => {
|
|
620
|
+
const capped = (): ModelConfig[] => [
|
|
621
|
+
{
|
|
622
|
+
alias: 'small', provider: ModelProvider.OpenAI, model: 'small-model', secret: 'sk-test',
|
|
623
|
+
maxTokens: 16000, maxTokensCap: 64000, maxOutput: 8000, contextWindow: 200_000,
|
|
624
|
+
},
|
|
625
|
+
{
|
|
626
|
+
alias: 'honest', provider: ModelProvider.OpenAI, model: 'honest-model', secret: 'sk-test',
|
|
627
|
+
maxTokens: 4000, maxTokensCap: 32000, maxOutput: 64000, contextWindow: 200_000,
|
|
628
|
+
fallback: { model: 'big-model', maxOutput: 128_000, contextWindow: 1_000_000 },
|
|
629
|
+
},
|
|
630
|
+
]
|
|
631
|
+
|
|
632
|
+
test('the declared cap chooses the ceiling and the capability trims it', () => {
|
|
633
|
+
expect(resolveOutputCap({ maxTokensCap: 32000 })).toBe(32000)
|
|
634
|
+
expect(resolveOutputCap({ maxOutput: 64000 })).toBe(64000)
|
|
635
|
+
expect(resolveOutputCap({ maxTokensCap: 64000, maxOutput: 8000 })).toBe(8000)
|
|
636
|
+
expect(resolveOutputCap({ maxTokensCap: 16000, maxOutput: 64000 })).toBe(16000)
|
|
637
|
+
expect(resolveOutputCap({})).toBe(DEFAULT_MAX_OUTPUT_CAP)
|
|
638
|
+
})
|
|
639
|
+
|
|
640
|
+
test('an initial budget above the provider capability is clamped at build time', () => {
|
|
641
|
+
const service = makeLlmService({ models: capped }, 'spec-cap-clamp')
|
|
642
|
+
const config = (service.getModel('small') as unknown as { metadata: { config: ModelConfig } })
|
|
643
|
+
.metadata.config
|
|
644
|
+
|
|
645
|
+
expect(config.maxTokens).toBe(8000)
|
|
646
|
+
})
|
|
647
|
+
|
|
648
|
+
test('a fallback carries its own capability rather than the primary\'s', () => {
|
|
649
|
+
const service = makeLlmService({ models: capped }, 'spec-cap-fallback')
|
|
650
|
+
const primary = service.getModel('honest')
|
|
651
|
+
const fallback = (primary as unknown as { __fallbackModel?: BaseChatModel }).__fallbackModel!
|
|
652
|
+
const config = (fallback as unknown as { metadata: { config: ModelConfig } }).metadata.config
|
|
653
|
+
|
|
654
|
+
expect(config.maxOutput).toBe(128_000)
|
|
655
|
+
expect(resolveOutputCap(config)).toBe(32000)
|
|
656
|
+
})
|
|
657
|
+
})
|
|
658
|
+
|
|
659
|
+
/**
|
|
660
|
+
* The models that took the sampling knobs away also think adaptively whether asked to or not,
|
|
661
|
+
* and that thinking is billed against the same `max_tokens` as the answer. A budget sized for
|
|
662
|
+
* the answer alone gets spent on reasoning, and the completion comes back with no text at all.
|
|
663
|
+
*/
|
|
664
|
+
describe('@owlmeans/llm — adaptive-thinking output budget', () => {
|
|
665
|
+
const maxTokensOf = (model: BaseChatModel): number | undefined =>
|
|
666
|
+
(model as unknown as { maxTokens?: number }).maxTokens
|
|
667
|
+
|
|
668
|
+
test('an always-reasoning model gets room for the reasoning AND the answer', () => {
|
|
669
|
+
const model = build(anthropicPlugin, { model: 'claude-sonnet-5', maxTokens: 8192 })
|
|
670
|
+
|
|
671
|
+
expect(maxTokensOf(model)).toBe(ADAPTIVE_MIN_MAX_TOKENS)
|
|
672
|
+
})
|
|
673
|
+
|
|
674
|
+
test('the floor never lowers a preset that asked for more', () => {
|
|
675
|
+
const model = build(anthropicPlugin, {
|
|
676
|
+
model: 'claude-sonnet-5', maxTokens: 64_000, maxTokensCap: 64_000, maxOutput: 128_000,
|
|
677
|
+
})
|
|
678
|
+
|
|
679
|
+
expect(maxTokensOf(model)).toBe(64_000)
|
|
680
|
+
})
|
|
681
|
+
|
|
682
|
+
/** Raising the floor past what the provider accepts turns a retryable empty into a 400. */
|
|
683
|
+
test('the floor is clamped to what the provider accepts', () => {
|
|
684
|
+
const model = build(anthropicPlugin, {
|
|
685
|
+
model: 'claude-sonnet-5', maxTokens: 4096, maxOutput: 8192,
|
|
686
|
+
})
|
|
687
|
+
|
|
688
|
+
expect(maxTokensOf(model)).toBe(8192)
|
|
689
|
+
})
|
|
690
|
+
|
|
691
|
+
test('models that still accept sampling keep the budget their preset asked for', () => {
|
|
692
|
+
const model = build(anthropicPlugin, { model: 'claude-haiku-4-5', maxTokens: 8192 })
|
|
693
|
+
|
|
694
|
+
expect(maxTokensOf(model)).toBe(8192)
|
|
695
|
+
})
|
|
696
|
+
})
|
|
697
|
+
|
|
698
|
+
describe('@owlmeans/llm — anthropic thinking control', () => {
|
|
699
|
+
// `thinkingExplicitlySet` is what decides whether langchain puts `thinking` on the wire at
|
|
700
|
+
// all; the field itself defaults to `disabled` and so proves nothing on its own.
|
|
701
|
+
const wire = (model: unknown) => model as { thinkingExplicitlySet: boolean; thinking: { type: string } }
|
|
702
|
+
|
|
703
|
+
test('disableThinking sends an explicit thinking:disabled to the adaptive family', () => {
|
|
704
|
+
const on = wire(build(anthropicPlugin, { model: 'claude-sonnet-5', disableThinking: true }))
|
|
705
|
+
expect(on.thinkingExplicitlySet).toBe(true)
|
|
706
|
+
expect(on.thinking).toEqual({ type: 'disabled' })
|
|
707
|
+
expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-sonnet-5', disableThinking: true } as ModelConfig)).toBe(true)
|
|
708
|
+
})
|
|
709
|
+
|
|
710
|
+
test('without the flag nothing is sent and the provider default applies', () => {
|
|
711
|
+
expect(wire(build(anthropicPlugin, { model: 'claude-sonnet-5' })).thinkingExplicitlySet).toBe(false)
|
|
712
|
+
expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-sonnet-5' } as ModelConfig)).toBe(false)
|
|
713
|
+
})
|
|
714
|
+
|
|
715
|
+
test('models that reason only when asked are left alone', () => {
|
|
716
|
+
expect(wire(build(anthropicPlugin, { model: 'claude-haiku-4-5', disableThinking: true })).thinkingExplicitlySet).toBe(false)
|
|
717
|
+
expect(anthropicPlugin.suppressesThinking?.({ alias: 'a', model: 'claude-haiku-4-5', disableThinking: true } as ModelConfig)).toBe(false)
|
|
718
|
+
})
|
|
719
|
+
|
|
720
|
+
test('refine keeps thinking off on every attempt', () => {
|
|
721
|
+
const base = build(anthropicPlugin, { model: 'claude-sonnet-5', disableThinking: true, maxTokens: 1000 })
|
|
722
|
+
const refined = wire(anthropicPlugin.refine({ base, attempt: 1, maxOutputCap: 64000 }))
|
|
723
|
+
expect(refined.thinkingExplicitlySet).toBe(true)
|
|
724
|
+
expect(refined.thinking).toEqual({ type: 'disabled' })
|
|
725
|
+
})
|
|
726
|
+
})
|
package/tests/prompt.spec.ts
CHANGED
|
@@ -148,6 +148,49 @@ describe('@owlmeans/llm — prompt composition', () => {
|
|
|
148
148
|
expect(context[0]!.text).toContain('second note')
|
|
149
149
|
})
|
|
150
150
|
|
|
151
|
+
// Two plugins can each be able to render the same skill — a static catalogue and a
|
|
152
|
+
// detector. Without the claim they both do, and the model is told the same thing twice.
|
|
153
|
+
test('a key is claimed by the first plugin only, within one composition', async () => {
|
|
154
|
+
const grants: boolean[] = []
|
|
155
|
+
const claimer = (alias: string): LlmPromptPlugin => ({
|
|
156
|
+
alias,
|
|
157
|
+
compose: ctx => {
|
|
158
|
+
if (ctx.claim('skill:a')) {
|
|
159
|
+
ctx.add(PromptBlock.Packages, `from ${alias}`)
|
|
160
|
+
}
|
|
161
|
+
grants.push(ctx.claim('probe'))
|
|
162
|
+
},
|
|
163
|
+
})
|
|
164
|
+
const result = await compose(service([], [claimer('first'), claimer('second')]))
|
|
165
|
+
|
|
166
|
+
expect(grants).toEqual([true, false])
|
|
167
|
+
expect(result.blocks.find(b => b.block === PromptBlock.Packages)?.text).toBe('from first')
|
|
168
|
+
})
|
|
169
|
+
|
|
170
|
+
test('the claim set is per composition, not per service', async () => {
|
|
171
|
+
const svc = service([], [{
|
|
172
|
+
alias: 'claimer', compose: ctx => ctx.add(PromptBlock.Packages, `${ctx.claim('once')}`),
|
|
173
|
+
}])
|
|
174
|
+
const first = await compose(svc)
|
|
175
|
+
const second = await compose(svc)
|
|
176
|
+
|
|
177
|
+
expect(first.blocks[0]!.text).toBe('true')
|
|
178
|
+
expect(second.blocks[0]!.text).toBe('true')
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
// The seam has to be free: a composition where nobody claims must render the bytes it
|
|
182
|
+
// rendered before the seam existed, or every cached prefix in the fleet is invalidated.
|
|
183
|
+
test('claiming nothing changes nothing about the composed bytes', async () => {
|
|
184
|
+
const input = { role: 'R', skills: ['a'], context: ['note'] }
|
|
185
|
+
const plain = await compose(service([skill('a', 'A')]), input)
|
|
186
|
+
const claiming = await compose(
|
|
187
|
+
service([skill('a', 'A')], [{ alias: 'quiet', compose: ctx => { ctx.claim('a') } }]),
|
|
188
|
+
input,
|
|
189
|
+
)
|
|
190
|
+
expect(JSON.stringify(claiming.system)).toBe(JSON.stringify(plain.system))
|
|
191
|
+
expect(claiming.breakpoints).toBe(plain.breakpoints)
|
|
192
|
+
})
|
|
193
|
+
|
|
151
194
|
test('nothing declared composes to no system message at all', async () => {
|
|
152
195
|
const result = await compose(service())
|
|
153
196
|
expect(result.system).toBeNull()
|