@owlmeans/llm 0.1.15 → 0.1.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -4
- package/agent-meta/manifest.json +9 -9
- package/agent-meta/skills/llm/SKILL.md +54 -4
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
- package/build/consts.d.ts +36 -3
- package/build/consts.d.ts.map +1 -1
- package/build/consts.js +37 -4
- package/build/consts.js.map +1 -1
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +8 -3
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +20 -1
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts +12 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +22 -0
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/cache.d.ts +17 -0
- package/build/helpers/cache.d.ts.map +1 -0
- package/build/helpers/cache.js +22 -0
- package/build/helpers/cache.js.map +1 -0
- package/build/helpers/index.d.ts +1 -0
- package/build/helpers/index.d.ts.map +1 -1
- package/build/helpers/index.js +1 -0
- package/build/helpers/index.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +7 -0
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +1 -0
- package/build/index.d.ts.map +1 -1
- package/build/index.js +1 -0
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +90 -16
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +136 -21
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +9 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +23 -1
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +38 -5
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/index.d.ts +5 -0
- package/build/prompt/index.d.ts.map +1 -0
- package/build/prompt/index.js +4 -0
- package/build/prompt/index.js.map +1 -0
- package/build/prompt/plugins.d.ts +24 -0
- package/build/prompt/plugins.d.ts.map +1 -0
- package/build/prompt/plugins.js +64 -0
- package/build/prompt/plugins.js.map +1 -0
- package/build/prompt/render.d.ts +28 -0
- package/build/prompt/render.d.ts.map +1 -0
- package/build/prompt/render.js +39 -0
- package/build/prompt/render.js.map +1 -0
- package/build/prompt/service.d.ts +16 -0
- package/build/prompt/service.d.ts.map +1 -0
- package/build/prompt/service.js +145 -0
- package/build/prompt/service.js.map +1 -0
- package/build/prompt/types.d.ts +101 -0
- package/build/prompt/types.d.ts.map +1 -0
- package/build/prompt/types.js +2 -0
- package/build/prompt/types.js.map +1 -0
- package/build/service.d.ts.map +1 -1
- package/build/service.js +3 -0
- package/build/service.js.map +1 -1
- package/build/types.d.ts +53 -3
- package/build/types.d.ts.map +1 -1
- package/build/utils/prompt.d.ts +14 -0
- package/build/utils/prompt.d.ts.map +1 -1
- package/build/utils/prompt.js +32 -0
- package/build/utils/prompt.js.map +1 -1
- package/package.json +13 -6
- package/src/consts.ts +43 -5
- package/src/execution/service.ts +9 -3
- package/src/execution/types.ts +21 -2
- package/src/execution/utils.ts +28 -1
- package/src/helpers/cache.ts +33 -0
- package/src/helpers/index.ts +1 -0
- package/src/helpers/spectate.ts +10 -0
- package/src/index.ts +1 -0
- package/src/model.ts +119 -16
- package/src/plugins/anthropic.ts +159 -20
- package/src/plugins/openai.ts +26 -1
- package/src/plugins/types.ts +42 -5
- package/src/prompt/index.ts +5 -0
- package/src/prompt/plugins.ts +69 -0
- package/src/prompt/render.ts +48 -0
- package/src/prompt/service.ts +194 -0
- package/src/prompt/types.ts +114 -0
- package/src/service.ts +3 -0
- package/src/types.ts +54 -3
- package/src/utils/prompt.ts +33 -0
- package/tests/execution.spec.ts +42 -0
- package/tests/plugins.spec.ts +279 -14
- package/tests/prompt.spec.ts +194 -0
- package/agent-meta/instructions/llm.instructions.md +0 -66
package/src/utils/prompt.ts
CHANGED
|
@@ -23,6 +23,39 @@ const appendDirective = (msgs: MessageFieldWithRole[], text: string): void => {
|
|
|
23
23
|
}
|
|
24
24
|
}
|
|
25
25
|
|
|
26
|
+
/**
|
|
27
|
+
* Drop every `cache_control` marker from the messages.
|
|
28
|
+
*
|
|
29
|
+
* Markers are placed in-place, on the caller's own message objects — and a caller that
|
|
30
|
+
* carries its message array across calls (the coder's growing conversation, a fix loop
|
|
31
|
+
* re-sending the same sources) hands them back still marked. The provider counts markers
|
|
32
|
+
* per REQUEST, not per message: Anthropic rejects the fifth outright with
|
|
33
|
+
* `400 A maximum of 4 blocks with cache_control may be provided. Found 5.`, and since a
|
|
34
|
+
* 400 is fatal it burns the entire retry budget before surfacing.
|
|
35
|
+
*
|
|
36
|
+
* So the pipeline always starts from a clean slate and re-places its own markers, which
|
|
37
|
+
* makes the per-request count a function of THIS call alone.
|
|
38
|
+
*/
|
|
39
|
+
export const stripCacheMarkers = (msgs: MessageFieldWithRole[]): void => {
|
|
40
|
+
for (const msg of msgs) {
|
|
41
|
+
if (!Array.isArray(msg.content)) {
|
|
42
|
+
continue
|
|
43
|
+
}
|
|
44
|
+
let found = false
|
|
45
|
+
const blocks = msg.content.map(part => {
|
|
46
|
+
if (typeof part === 'object' && part !== null && 'cache_control' in part) {
|
|
47
|
+
found = true
|
|
48
|
+
const { cache_control: _dropped, ...rest } = part as Record<string, unknown>
|
|
49
|
+
return rest
|
|
50
|
+
}
|
|
51
|
+
return part
|
|
52
|
+
})
|
|
53
|
+
if (found) {
|
|
54
|
+
msg.content = blocks as unknown as typeof msg.content
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
26
59
|
/**
|
|
27
60
|
* Ensure the word "json" appears somewhere in the prompt. Several providers refuse or
|
|
28
61
|
* silently ignore a JSON mode unless it does; when it is missing the JSON instruction is
|
package/tests/execution.spec.ts
CHANGED
|
@@ -180,6 +180,48 @@ describe('@owlmeans/llm — snapshot and restore', () => {
|
|
|
180
180
|
})
|
|
181
181
|
})
|
|
182
182
|
|
|
183
|
+
describe('@owlmeans/llm — prompt policy accumulation', () => {
|
|
184
|
+
const withPrompt = () => service.root({
|
|
185
|
+
models: makeService(),
|
|
186
|
+
policy: { effort: DEFAULT_EFFORT },
|
|
187
|
+
purpose: { type: 'spec' },
|
|
188
|
+
prompt: { role: 'project role', skills: ['base'] },
|
|
189
|
+
})
|
|
190
|
+
|
|
191
|
+
// Skills accumulate as work narrows — that is how a helper ends up knowing everything
|
|
192
|
+
// the project, the task and its own role declared, without any of them repeating it.
|
|
193
|
+
test('skills accumulate down the chain while the deepest role wins', () => {
|
|
194
|
+
const task = service.forTask(withPrompt(), { prompt: { skills: ['task'] } })
|
|
195
|
+
expect(task.prompt?.skills).toEqual(['base', 'task'])
|
|
196
|
+
expect(task.prompt?.role).toBe('project role')
|
|
197
|
+
|
|
198
|
+
const helper = service.forHelper(task, {
|
|
199
|
+
role: Role.Analyst, prompt: { role: 'helper role', skills: ['helper'] },
|
|
200
|
+
})
|
|
201
|
+
expect(helper.prompt?.skills).toEqual(['base', 'task', 'helper'])
|
|
202
|
+
expect(helper.prompt?.role).toBe('helper role')
|
|
203
|
+
})
|
|
204
|
+
|
|
205
|
+
// A skill declared twice must not render twice, or the composed prefix differs from the
|
|
206
|
+
// one a single declaration would have produced.
|
|
207
|
+
test('a repeated skill is unioned, not duplicated', () => {
|
|
208
|
+
const task = service.forTask(withPrompt(), { prompt: { skills: ['base', 'task'] } })
|
|
209
|
+
expect(task.prompt?.skills).toEqual(['base', 'task'])
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
test('levels that declare nothing inherit the policy untouched', () => {
|
|
213
|
+
const helper = service.forHelper(service.forTask(withPrompt(), {}), { role: Role.Analyst })
|
|
214
|
+
expect(helper.prompt).toEqual({ role: 'project role', skills: ['base'] })
|
|
215
|
+
})
|
|
216
|
+
|
|
217
|
+
// It is part of ExecutionState, so a resumed run rebuilds the same system prompt.
|
|
218
|
+
test('the policy survives a snapshot/restore round trip', () => {
|
|
219
|
+
const helper = service.forHelper(withPrompt(), { role: Role.Analyst })
|
|
220
|
+
const restored = service.restore(service.snapshot(helper), { models: makeService() })
|
|
221
|
+
expect(restored.prompt).toEqual({ role: 'project role', skills: ['base'] })
|
|
222
|
+
})
|
|
223
|
+
})
|
|
224
|
+
|
|
183
225
|
describe('@owlmeans/llm — resilience plugin seam', () => {
|
|
184
226
|
test('checkpoint is a no-op until a plugin is registered', async () => {
|
|
185
227
|
await expect(service.checkpoint(root, 'key')).resolves.toBeUndefined()
|
package/tests/plugins.spec.ts
CHANGED
|
@@ -2,13 +2,14 @@ import { describe, expect, test } from 'bun:test'
|
|
|
2
2
|
import { ChatAnthropic } from '@langchain/anthropic'
|
|
3
3
|
import { ChatOpenAI } from '@langchain/openai'
|
|
4
4
|
import { BadRequestError } from '@anthropic-ai/sdk'
|
|
5
|
-
import { ModelProvider, StructuredMode } from '@owlmeans/llm-common'
|
|
5
|
+
import { ModelProvider, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
|
|
6
6
|
import {
|
|
7
7
|
anthropicPlugin, compatiblePlugin, makeLlmService, openAiPlugin, pluginFor, pluginOf,
|
|
8
8
|
registerLlmPlugin, resolvePlugin,
|
|
9
9
|
} from '@owlmeans/llm'
|
|
10
10
|
import type { LlmPlugin, ModelConfig } from '@owlmeans/llm'
|
|
11
11
|
import { offlineConfigs, Role } from './context.js'
|
|
12
|
+
import { stripCacheMarkers } from '../src/utils/prompt.js'
|
|
12
13
|
|
|
13
14
|
const build = (plugin: LlmPlugin, config: Partial<ModelConfig> = {}) =>
|
|
14
15
|
plugin.build({
|
|
@@ -164,26 +165,251 @@ describe('@owlmeans/llm — retry escalation behaviour', () => {
|
|
|
164
165
|
})
|
|
165
166
|
})
|
|
166
167
|
|
|
167
|
-
describe('@owlmeans/llm — prompt caching', () => {
|
|
168
|
-
|
|
168
|
+
describe('@owlmeans/llm — message prompt caching', () => {
|
|
169
|
+
/** `cacheMinTokens: 1` puts the minimum at 4 characters so fixtures stay readable. */
|
|
170
|
+
const cheap = () => build(anthropicPlugin, { model: 'claude-haiku-4-5-20251001', cacheMinTokens: 1 })
|
|
171
|
+
|
|
172
|
+
// The budget is four breakpoints for the WHOLE request, and the composed system prompt
|
|
173
|
+
// has first claim on it — so the messages get one marker at the end of their stable
|
|
174
|
+
// prefix, not one marker each.
|
|
175
|
+
test('anthropic places a single breakpoint at the end of the stable prefix', () => {
|
|
176
|
+
const msgs = [
|
|
177
|
+
{ role: 'system' as const, content: 'aaaa' },
|
|
178
|
+
{ role: 'user' as const, content: 'bbbb' },
|
|
179
|
+
{ role: 'user' as const, content: 'cccc' },
|
|
180
|
+
]
|
|
181
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 2 })).toBe(true)
|
|
182
|
+
expect(msgs[0]!.content).toBe('aaaa')
|
|
183
|
+
expect(msgs[1]!.content).toEqual([{ type: 'text', text: 'bbbb', cache_control: { type: 'ephemeral' } }])
|
|
184
|
+
expect(msgs[2]!.content).toBe('cccc')
|
|
185
|
+
})
|
|
186
|
+
|
|
187
|
+
// The last message is the per-call payload (and `ensureJsonMention` / `applyNoThink`
|
|
188
|
+
// append to it) — caching it would write a fresh entry every call and read none.
|
|
189
|
+
test('the final message is never part of the cached prefix', () => {
|
|
190
|
+
const msgs = [
|
|
191
|
+
{ role: 'system' as const, content: 'aaaa' },
|
|
192
|
+
{ role: 'user' as const, content: 'bbbb' },
|
|
193
|
+
]
|
|
194
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 9 })).toBe(true)
|
|
195
|
+
expect(msgs[0]!.content).toEqual([{ type: 'text', text: 'aaaa', cache_control: { type: 'ephemeral' } }])
|
|
196
|
+
expect(msgs[1]!.content).toBe('bbbb')
|
|
197
|
+
|
|
198
|
+
const single = [{ role: 'user' as const, content: 'aaaa' }]
|
|
199
|
+
expect(anthropicPlugin.patchCache?.(single, { model: cheap(), useCache: true, cacheMax: 4 })).toBe(false)
|
|
200
|
+
expect(single[0]!.content).toBe('aaaa')
|
|
201
|
+
})
|
|
202
|
+
|
|
203
|
+
test('a marker is appended to block content rather than replacing it, and is idempotent', () => {
|
|
204
|
+
const msgs = [
|
|
205
|
+
{ role: 'system' as const, content: [{ type: 'text', text: 'aa' }, { type: 'text', text: 'bb' }] },
|
|
206
|
+
{ role: 'user' as const, content: 'cccc' },
|
|
207
|
+
]
|
|
208
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 1 })).toBe(true)
|
|
209
|
+
expect(msgs[0]!.content).toEqual([
|
|
210
|
+
{ type: 'text', text: 'aa' },
|
|
211
|
+
{ type: 'text', text: 'bb', cache_control: { type: 'ephemeral' } },
|
|
212
|
+
])
|
|
213
|
+
|
|
214
|
+
const before = JSON.stringify(msgs[0]!.content)
|
|
215
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 1 })).toBe(true)
|
|
216
|
+
expect(JSON.stringify(msgs[0]!.content)).toBe(before)
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
// A prefix under the provider's own minimum is silently never cached, so a marker there
|
|
220
|
+
// buys nothing and costs one of the four breakpoints.
|
|
221
|
+
test('a prefix below the cacheable minimum is left unmarked', () => {
|
|
169
222
|
const model = build(anthropicPlugin, { model: 'claude-haiku-4-5-20251001' })
|
|
170
223
|
const msgs = [
|
|
171
|
-
{ role: 'system' as const, content: '
|
|
172
|
-
{ role: 'user' as const, content: '
|
|
173
|
-
{ role: 'user' as const, content: 'c' },
|
|
224
|
+
{ role: 'system' as const, content: 'short' },
|
|
225
|
+
{ role: 'user' as const, content: 'also short' },
|
|
174
226
|
]
|
|
175
|
-
expect(anthropicPlugin.patchCache?.(msgs, { model, useCache: true, cacheMax:
|
|
176
|
-
expect(msgs[0]!.content).
|
|
177
|
-
|
|
178
|
-
|
|
227
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model, useCache: true, cacheMax: 4 })).toBe(false)
|
|
228
|
+
expect(msgs[0]!.content).toBe('short')
|
|
229
|
+
})
|
|
230
|
+
|
|
231
|
+
test('the message budget yields to whatever the system prompt already spent', () => {
|
|
232
|
+
const msgs = [
|
|
233
|
+
{ role: 'system' as const, content: 'aaaa' },
|
|
234
|
+
{ role: 'user' as const, content: 'bbbb' },
|
|
235
|
+
]
|
|
236
|
+
expect(anthropicPlugin.patchCache?.(
|
|
237
|
+
msgs, { model: cheap(), useCache: true, cacheMax: 4, reserved: 4 }
|
|
238
|
+
)).toBe(false)
|
|
239
|
+
expect(msgs[0]!.content).toBe('aaaa')
|
|
179
240
|
})
|
|
180
241
|
|
|
181
242
|
test('caching is a no-op when not requested, and for providers without it', () => {
|
|
182
|
-
const
|
|
183
|
-
|
|
184
|
-
expect(
|
|
185
|
-
expect(msgs[0]!.content).toBe('a')
|
|
243
|
+
const msgs = [{ role: 'user' as const, content: 'aaaa' }, { role: 'user' as const, content: 'bbbb' }]
|
|
244
|
+
expect(anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: false, cacheMax: 4 })).toBe(false)
|
|
245
|
+
expect(msgs[0]!.content).toBe('aaaa')
|
|
186
246
|
expect(openAiPlugin.patchCache).toBeUndefined()
|
|
247
|
+
expect(openAiPlugin.patchSystem).toBeUndefined()
|
|
248
|
+
})
|
|
249
|
+
})
|
|
250
|
+
|
|
251
|
+
describe('@owlmeans/llm — system prompt caching', () => {
|
|
252
|
+
const model = () => build(anthropicPlugin, { model: 'claude-haiku-4-5-20251001', cacheMinTokens: 1 })
|
|
253
|
+
const blocks = (...pairs: Array<[PromptBlock, string]>) =>
|
|
254
|
+
pairs.map(([block, text]) => ({ block, text }))
|
|
255
|
+
|
|
256
|
+
const marks = (content: unknown): boolean[] =>
|
|
257
|
+
(content as Array<Record<string, unknown>>).map(part => part.cache_control != null)
|
|
258
|
+
|
|
259
|
+
// Role + skills is the region every call of this role shares; packages vary with the
|
|
260
|
+
// request. Two boundaries, so a changing package block can never invalidate the skills.
|
|
261
|
+
test('breakpoints land on the stability boundaries, not on every block', () => {
|
|
262
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
263
|
+
blocks(
|
|
264
|
+
[PromptBlock.Role, 'role text'],
|
|
265
|
+
[PromptBlock.Skills, 'skill text'],
|
|
266
|
+
[PromptBlock.Packages, 'package text'],
|
|
267
|
+
),
|
|
268
|
+
{ model: model(), cacheMax: 3, ttl: '5m' },
|
|
269
|
+
)
|
|
270
|
+
expect(render?.breakpoints).toBe(2)
|
|
271
|
+
expect(marks(render?.content)).toEqual([false, true, true])
|
|
272
|
+
})
|
|
273
|
+
|
|
274
|
+
// "Fully cached by default" also has to hold for a caller that has not migrated and
|
|
275
|
+
// still hands over a single system message of its own.
|
|
276
|
+
test('a lone context block is still cached', () => {
|
|
277
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
278
|
+
blocks([PromptBlock.Context, 'legacy system message']),
|
|
279
|
+
{ model: model(), cacheMax: 3, ttl: '5m' },
|
|
280
|
+
)
|
|
281
|
+
expect(render?.breakpoints).toBe(1)
|
|
282
|
+
expect(marks(render?.content)).toEqual([true])
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
test('marking stops at the budget, earliest boundary first', () => {
|
|
286
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
287
|
+
blocks(
|
|
288
|
+
[PromptBlock.Skills, 'skill text'],
|
|
289
|
+
[PromptBlock.Packages, 'package text'],
|
|
290
|
+
[PromptBlock.Context, 'context text'],
|
|
291
|
+
),
|
|
292
|
+
{ model: model(), cacheMax: 1, ttl: '5m' },
|
|
293
|
+
)
|
|
294
|
+
expect(render?.breakpoints).toBe(1)
|
|
295
|
+
expect(marks(render?.content)).toEqual([true, false, false])
|
|
296
|
+
})
|
|
297
|
+
|
|
298
|
+
// An explicit `"ttl": "5m"` is different BYTES from no ttl at all, and a prefix cached
|
|
299
|
+
// one way would not match the other.
|
|
300
|
+
test('the default ttl is omitted from the marker, and 1h is spelled out', () => {
|
|
301
|
+
const short = anthropicPlugin.patchSystem?.(
|
|
302
|
+
blocks([PromptBlock.Skills, 'skill text']), { model: model(), cacheMax: 3, ttl: '5m' }
|
|
303
|
+
)
|
|
304
|
+
expect((short?.content as Array<Record<string, unknown>>)[0]!.cache_control)
|
|
305
|
+
.toEqual({ type: 'ephemeral' })
|
|
306
|
+
|
|
307
|
+
const long = anthropicPlugin.patchSystem?.(
|
|
308
|
+
blocks([PromptBlock.Skills, 'skill text']), { model: model(), cacheMax: 3, ttl: '1h' }
|
|
309
|
+
)
|
|
310
|
+
expect((long?.content as Array<Record<string, unknown>>)[0]!.cache_control)
|
|
311
|
+
.toEqual({ type: 'ephemeral', ttl: '1h' })
|
|
312
|
+
})
|
|
313
|
+
|
|
314
|
+
// A trailing context block changes every call. Marking it would pay a cache WRITE per
|
|
315
|
+
// call and never read one back — a breakpoint spent to buy nothing.
|
|
316
|
+
test('a trailing context block is never marked when stable blocks precede it', () => {
|
|
317
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
318
|
+
blocks(
|
|
319
|
+
[PromptBlock.Role, 'role text'],
|
|
320
|
+
[PromptBlock.Skills, 'skill text'],
|
|
321
|
+
[PromptBlock.Context, 'per-call text'],
|
|
322
|
+
),
|
|
323
|
+
{ model: model(), cacheMax: 2, ttl: '5m' },
|
|
324
|
+
)
|
|
325
|
+
expect(render?.breakpoints).toBe(1)
|
|
326
|
+
expect(marks(render?.content)).toEqual([false, true, false])
|
|
327
|
+
})
|
|
328
|
+
|
|
329
|
+
test('the packages boundary is still marked with context trailing behind it', () => {
|
|
330
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
331
|
+
blocks(
|
|
332
|
+
[PromptBlock.Role, 'role text'],
|
|
333
|
+
[PromptBlock.Skills, 'skill text'],
|
|
334
|
+
[PromptBlock.Packages, 'package text'],
|
|
335
|
+
[PromptBlock.Context, 'per-call text'],
|
|
336
|
+
),
|
|
337
|
+
{ model: model(), cacheMax: 2, ttl: '5m' },
|
|
338
|
+
)
|
|
339
|
+
expect(render?.breakpoints).toBe(2)
|
|
340
|
+
expect(marks(render?.content)).toEqual([false, true, true, false])
|
|
341
|
+
})
|
|
342
|
+
|
|
343
|
+
test('a non-claude model still renders the blocks, just without markers', () => {
|
|
344
|
+
const render = anthropicPlugin.patchSystem?.(
|
|
345
|
+
blocks([PromptBlock.Role, 'role text']),
|
|
346
|
+
{ model: build(anthropicPlugin, { model: 'some-other-model' }), cacheMax: 3, ttl: '5m' },
|
|
347
|
+
)
|
|
348
|
+
expect(render?.breakpoints).toBe(0)
|
|
349
|
+
expect(render?.content).toEqual([{ type: 'text', text: 'role text' }])
|
|
350
|
+
})
|
|
351
|
+
})
|
|
352
|
+
|
|
353
|
+
describe('@owlmeans/llm — the four-breakpoint request budget', () => {
|
|
354
|
+
const cheap = () => build(anthropicPlugin, { model: 'claude-haiku-4-5-20251001', cacheMinTokens: 1 })
|
|
355
|
+
|
|
356
|
+
const markers = (msgs: Array<{ content: unknown }>): number =>
|
|
357
|
+
msgs.reduce((sum, msg) => sum + (Array.isArray(msg.content)
|
|
358
|
+
? (msg.content as Array<Record<string, unknown>>).filter(b => b.cache_control != null).length
|
|
359
|
+
: 0), 0)
|
|
360
|
+
|
|
361
|
+
// The live failure: a caller that carries its message array across calls hands back
|
|
362
|
+
// messages this pipeline already marked. They accumulate until Anthropic rejects the
|
|
363
|
+
// request with `400 A maximum of 4 blocks with cache_control may be provided. Found 5.`
|
|
364
|
+
test('markers left over from a previous call are cleared before new ones are placed', () => {
|
|
365
|
+
const msgs = [
|
|
366
|
+
{ role: 'system' as const, content: 'aaaa' },
|
|
367
|
+
{ role: 'user' as const, content: 'bbbb' },
|
|
368
|
+
{ role: 'user' as const, content: 'cccc' },
|
|
369
|
+
]
|
|
370
|
+
// Call one marks the stable prefix.
|
|
371
|
+
anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 2 })
|
|
372
|
+
expect(markers(msgs)).toBe(1)
|
|
373
|
+
|
|
374
|
+
// Call two: the caller appends a turn and re-sends the SAME objects.
|
|
375
|
+
msgs.push({ role: 'user' as const, content: 'dddd' })
|
|
376
|
+
stripCacheMarkers(msgs)
|
|
377
|
+
expect(markers(msgs)).toBe(0)
|
|
378
|
+
|
|
379
|
+
anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 3 })
|
|
380
|
+
expect(markers(msgs)).toBe(1)
|
|
381
|
+
})
|
|
382
|
+
|
|
383
|
+
test('the system prompt plus the message prefix never exceed the provider limit', () => {
|
|
384
|
+
// Worst case: every stability boundary distinct, so the system claims its full share.
|
|
385
|
+
const system = anthropicPlugin.patchSystem?.(
|
|
386
|
+
[
|
|
387
|
+
{ block: PromptBlock.Role, text: 'role text' },
|
|
388
|
+
{ block: PromptBlock.Skills, text: 'skill text' },
|
|
389
|
+
{ block: PromptBlock.Packages, text: 'package text' },
|
|
390
|
+
{ block: PromptBlock.Context, text: 'context text' },
|
|
391
|
+
],
|
|
392
|
+
{ model: cheap(), cacheMax: 3, ttl: '5m' },
|
|
393
|
+
)
|
|
394
|
+
const reserved = system?.breakpoints ?? 0
|
|
395
|
+
expect(reserved).toBeLessThanOrEqual(3)
|
|
396
|
+
|
|
397
|
+
const msgs = [
|
|
398
|
+
{ role: 'system' as const, content: system?.content as never },
|
|
399
|
+
{ role: 'user' as const, content: 'bbbb' },
|
|
400
|
+
{ role: 'user' as const, content: 'cccc' },
|
|
401
|
+
]
|
|
402
|
+
anthropicPlugin.patchCache?.(msgs, { model: cheap(), useCache: true, cacheMax: 4, reserved })
|
|
403
|
+
expect(markers(msgs)).toBeLessThanOrEqual(4)
|
|
404
|
+
})
|
|
405
|
+
|
|
406
|
+
test('stripping leaves the rest of a content block untouched', () => {
|
|
407
|
+
const msgs = [{
|
|
408
|
+
role: 'user' as const,
|
|
409
|
+
content: [{ type: 'text', text: 'keep me', cache_control: { type: 'ephemeral' } }] as never,
|
|
410
|
+
}]
|
|
411
|
+
stripCacheMarkers(msgs)
|
|
412
|
+
expect(msgs[0]!.content).toEqual([{ type: 'text', text: 'keep me' }] as never)
|
|
187
413
|
})
|
|
188
414
|
})
|
|
189
415
|
|
|
@@ -193,6 +419,22 @@ describe('@owlmeans/llm — fatal error classification', () => {
|
|
|
193
419
|
expect(anthropicPlugin.isFatal?.(bad)).toBe(bad)
|
|
194
420
|
expect(anthropicPlugin.isFatal?.(new Error('transient'))).toBeNull()
|
|
195
421
|
})
|
|
422
|
+
|
|
423
|
+
// `@langchain/anthropic` bundles its OWN nested copy of `@anthropic-ai/sdk`, so the
|
|
424
|
+
// error it throws is an instance of a different class than the one imported here and
|
|
425
|
+
// `instanceof` silently misses. That turned every fatal 400 into eight full retries.
|
|
426
|
+
test('a 400 from a foreign SDK copy is still fatal', () => {
|
|
427
|
+
const foreign = Object.assign(new Error('400 too many cache_control blocks'), { status: 400 })
|
|
428
|
+
expect(anthropicPlugin.isFatal?.(foreign)).toBe(foreign)
|
|
429
|
+
expect(openAiPlugin.isFatal?.(foreign)).toBe(foreign)
|
|
430
|
+
expect(compatiblePlugin.isFatal?.(foreign)).toBe(foreign)
|
|
431
|
+
})
|
|
432
|
+
|
|
433
|
+
test('a retryable status is not treated as fatal', () => {
|
|
434
|
+
const overloaded = Object.assign(new Error('529'), { status: 529 })
|
|
435
|
+
expect(anthropicPlugin.isFatal?.(overloaded)).toBeNull()
|
|
436
|
+
expect(openAiPlugin.isFatal?.(overloaded)).toBeNull()
|
|
437
|
+
})
|
|
196
438
|
})
|
|
197
439
|
|
|
198
440
|
describe('@owlmeans/llm — service', () => {
|
|
@@ -203,6 +445,29 @@ describe('@owlmeans/llm — service', () => {
|
|
|
203
445
|
expect(service.getModel(Role.Analyst, {}, true)).not.toBe(first)
|
|
204
446
|
})
|
|
205
447
|
|
|
448
|
+
// The knob an application sets where it composes its context: one place to bound how
|
|
449
|
+
// long the whole deployment waits on a silent provider.
|
|
450
|
+
test('a service-wide idle deadline reaches every model it builds', () => {
|
|
451
|
+
const service = makeLlmService(
|
|
452
|
+
{ models: offlineConfigs, streamTimeout: 90_000 }, 'spec-llm-timeout'
|
|
453
|
+
)
|
|
454
|
+
const config = (service.getModel(Role.Analyst) as unknown as {
|
|
455
|
+
metadata: { config: ModelConfig }
|
|
456
|
+
}).metadata.config
|
|
457
|
+
expect(config.streamTimeout).toBe(90_000)
|
|
458
|
+
})
|
|
459
|
+
|
|
460
|
+
// A preset that names its own deadline knows something specific about that model.
|
|
461
|
+
test('a model config keeps its own deadline over the service default', () => {
|
|
462
|
+
const configs = (): ModelConfig[] =>
|
|
463
|
+
offlineConfigs().map(c => c.alias === Role.Analyst ? { ...c, streamTimeout: 12_000 } : c)
|
|
464
|
+
const service = makeLlmService({ models: configs, streamTimeout: 90_000 }, 'spec-llm-timeout-2')
|
|
465
|
+
const config = (service.getModel(Role.Analyst) as unknown as {
|
|
466
|
+
metadata: { config: ModelConfig }
|
|
467
|
+
}).metadata.config
|
|
468
|
+
expect(config.streamTimeout).toBe(12_000)
|
|
469
|
+
})
|
|
470
|
+
|
|
206
471
|
test('an override participates in the cache key', () => {
|
|
207
472
|
const service = makeLlmService({ models: offlineConfigs }, 'spec-llm-override')
|
|
208
473
|
const plain = service.getModel(Role.Analyst)
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
import { describe, expect, test } from 'bun:test'
|
|
2
|
+
import { PromptBlock } from '@owlmeans/llm-common'
|
|
3
|
+
import type { SkillDefinition } from '@owlmeans/llm-common'
|
|
4
|
+
import { anthropicPlugin, makePromptService } from '@owlmeans/llm'
|
|
5
|
+
import type { LlmPromptPlugin, ModelConfig, PromptService } from '@owlmeans/llm'
|
|
6
|
+
|
|
7
|
+
/** `cacheMinTokens: 1` puts the cacheable minimum at 4 characters so fixtures stay short. */
|
|
8
|
+
const model = anthropicPlugin.build({
|
|
9
|
+
alias: 'spec',
|
|
10
|
+
secret: 'sk-test',
|
|
11
|
+
callbacks: [],
|
|
12
|
+
config: { alias: 'spec', model: 'claude-haiku-4-5-20251001', cacheMinTokens: 1 } as ModelConfig,
|
|
13
|
+
})
|
|
14
|
+
|
|
15
|
+
let seq = 0
|
|
16
|
+
const service = (skills: SkillDefinition[] = [], plugins: LlmPromptPlugin[] = []): PromptService =>
|
|
17
|
+
makePromptService({ skills, plugins }, `spec-prompt-${seq++}`)
|
|
18
|
+
|
|
19
|
+
const compose = (svc: PromptService, input: Parameters<PromptService['compose']>[0] = {}) =>
|
|
20
|
+
svc.compose(input, [{ role: 'user', content: 'the task' }], { model, provider: anthropicPlugin })
|
|
21
|
+
|
|
22
|
+
const skill = (alias: string, body: string, extra: Partial<SkillDefinition> = {}): SkillDefinition =>
|
|
23
|
+
({ alias, body, ...extra })
|
|
24
|
+
|
|
25
|
+
describe('@owlmeans/llm — skill registry', () => {
|
|
26
|
+
test('resolve follows requires depth-first and de-duplicates', () => {
|
|
27
|
+
const svc = service([
|
|
28
|
+
skill('a', 'A', { requires: ['b', 'c'] }),
|
|
29
|
+
skill('b', 'B', { requires: ['c'] }),
|
|
30
|
+
skill('c', 'C'),
|
|
31
|
+
])
|
|
32
|
+
expect(svc.resolve(['a']).map(s => s.alias)).toEqual(['c', 'b', 'a'])
|
|
33
|
+
expect(svc.resolve(['a', 'b', 'c']).map(s => s.alias)).toEqual(['c', 'b', 'a'])
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
// A catalogue is assembled from several packages; a missing optional entry should
|
|
37
|
+
// degrade the prompt, not break the call.
|
|
38
|
+
test('an unknown alias is skipped rather than thrown', () => {
|
|
39
|
+
const svc = service([skill('known', 'K')])
|
|
40
|
+
expect(svc.resolve(['known', 'missing']).map(s => s.alias)).toEqual(['known'])
|
|
41
|
+
expect(svc.has('missing')).toBe(false)
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
test('a cyclic requires graph terminates', () => {
|
|
45
|
+
const svc = service([skill('a', 'A', { requires: ['b'] }), skill('b', 'B', { requires: ['a'] })])
|
|
46
|
+
expect(svc.resolve(['a']).map(s => s.alias).sort()).toEqual(['a', 'b'])
|
|
47
|
+
})
|
|
48
|
+
})
|
|
49
|
+
|
|
50
|
+
describe('@owlmeans/llm — prompt composition', () => {
|
|
51
|
+
test('blocks are emitted in stability order regardless of who contributed them', async () => {
|
|
52
|
+
const late: LlmPromptPlugin = {
|
|
53
|
+
alias: 'late', order: 99, compose: ctx => ctx.add(PromptBlock.Packages, 'package text'),
|
|
54
|
+
}
|
|
55
|
+
const result = await compose(
|
|
56
|
+
service([skill('s', 'skill text')], [late]),
|
|
57
|
+
{ role: 'role text', skills: ['s'], context: ['context text'] },
|
|
58
|
+
)
|
|
59
|
+
expect(result.blocks.map(block => block.block))
|
|
60
|
+
.toEqual([PromptBlock.Role, PromptBlock.Skills, PromptBlock.Packages, PromptBlock.Context])
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
// The whole design rests on this: a prompt cache is a byte-exact prefix match, so two
|
|
64
|
+
// calls that declare the same thing must render the same thing.
|
|
65
|
+
test('the same declaration composes to identical bytes', async () => {
|
|
66
|
+
const svc = service([skill('a', 'A'), skill('b', 'B')])
|
|
67
|
+
const first = await compose(svc, { role: 'R', skills: ['a', 'b'] })
|
|
68
|
+
const second = await compose(svc, { role: 'R', skills: ['b', 'a'] })
|
|
69
|
+
expect(JSON.stringify(second.system)).toBe(JSON.stringify(first.system))
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
test('skill order comes from the definitions, not from registration or request order', async () => {
|
|
73
|
+
const forward = service([skill('z', 'Z', { order: 1 }), skill('a', 'A', { order: 2 })])
|
|
74
|
+
const backward = service([skill('a', 'A', { order: 2 }), skill('z', 'Z', { order: 1 })])
|
|
75
|
+
const one = await compose(forward, { skills: ['a', 'z'] })
|
|
76
|
+
const two = await compose(backward, { skills: ['z', 'a'] })
|
|
77
|
+
expect(one.blocks[0]!.text).toBe(two.blocks[0]!.text)
|
|
78
|
+
expect(one.blocks[0]!.text.indexOf('## z')).toBeLessThan(one.blocks[0]!.text.indexOf('## a'))
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
// The reason packages get their own block: whatever a request happens to mention must
|
|
82
|
+
// not shift a single byte of the region every call shares.
|
|
83
|
+
test('a changing packages block leaves the cached region byte-identical', async () => {
|
|
84
|
+
const inject = (text: string): LlmPromptPlugin =>
|
|
85
|
+
({ alias: 'pkg', inspect: ctx => ctx.add(PromptBlock.Packages, text) })
|
|
86
|
+
const base = { role: 'R', skills: ['a'] }
|
|
87
|
+
const one = await compose(service([skill('a', 'A')], [inject('first')]), base)
|
|
88
|
+
const two = await compose(service([skill('a', 'A')], [inject('second')]), base)
|
|
89
|
+
|
|
90
|
+
const stable = (blocks: typeof one.blocks) =>
|
|
91
|
+
blocks.filter(b => b.block !== PromptBlock.Packages).map(b => b.text).join('|')
|
|
92
|
+
expect(stable(two.blocks)).toBe(stable(one.blocks))
|
|
93
|
+
expect(two.blocks.find(b => b.block === PromptBlock.Packages)?.text).toBe('second')
|
|
94
|
+
})
|
|
95
|
+
|
|
96
|
+
test('per-call skills render into the volatile context block, not the cached one', async () => {
|
|
97
|
+
const result = await compose(
|
|
98
|
+
service([skill('static', 'S'), skill('percall', 'P')]),
|
|
99
|
+
{ skills: ['static'], callSkills: ['percall'] },
|
|
100
|
+
)
|
|
101
|
+
expect(result.blocks.find(b => b.block === PromptBlock.Skills)?.text).toContain('## static')
|
|
102
|
+
expect(result.blocks.find(b => b.block === PromptBlock.Skills)?.text).not.toContain('## percall')
|
|
103
|
+
expect(result.blocks.find(b => b.block === PromptBlock.Context)?.text).toContain('## percall')
|
|
104
|
+
})
|
|
105
|
+
|
|
106
|
+
test('an inline skill overrides the registered one of the same alias', async () => {
|
|
107
|
+
const result = await compose(
|
|
108
|
+
service([skill('a', 'registered body')]),
|
|
109
|
+
{ skills: ['a'], inline: [skill('a', 'inline body')] },
|
|
110
|
+
)
|
|
111
|
+
expect(result.blocks[0]!.text).toContain('inline body')
|
|
112
|
+
expect(result.blocks[0]!.text).not.toContain('registered body')
|
|
113
|
+
})
|
|
114
|
+
|
|
115
|
+
test('registering the same plugin alias twice replaces it instead of emitting twice', async () => {
|
|
116
|
+
const svc = service()
|
|
117
|
+
const plugin: LlmPromptPlugin = {
|
|
118
|
+
alias: 'dup', compose: ctx => ctx.add(PromptBlock.Skills, 'once'),
|
|
119
|
+
}
|
|
120
|
+
svc.use(plugin)
|
|
121
|
+
svc.use(plugin)
|
|
122
|
+
const result = await compose(svc)
|
|
123
|
+
expect(result.blocks[0]!.text).toBe('once')
|
|
124
|
+
})
|
|
125
|
+
|
|
126
|
+
// Detection plugins need every static contribution already in place.
|
|
127
|
+
test('every compose pass runs before the first inspect pass', async () => {
|
|
128
|
+
const seen: string[] = []
|
|
129
|
+
const plugins: LlmPromptPlugin[] = [
|
|
130
|
+
{ alias: 'x', order: 1, compose: () => { seen.push('compose:x') }, inspect: () => { seen.push('inspect:x') } },
|
|
131
|
+
{ alias: 'y', order: 2, compose: () => { seen.push('compose:y') } },
|
|
132
|
+
]
|
|
133
|
+
await compose(service([], plugins))
|
|
134
|
+
expect(seen).toEqual(['compose:x', 'compose:y', 'inspect:x'])
|
|
135
|
+
})
|
|
136
|
+
|
|
137
|
+
// Volatile parts are merged rather than emitted separately: the block carries no
|
|
138
|
+
// breakpoint, so keeping them separable buys nothing and reads worse to the model.
|
|
139
|
+
test('every volatile part lands in a single context chunk', async () => {
|
|
140
|
+
const result = await compose(
|
|
141
|
+
service([skill('a', 'A'), skill('percall', 'P')]),
|
|
142
|
+
{ skills: ['a'], callSkills: ['percall'], context: ['first note', 'second note'] },
|
|
143
|
+
)
|
|
144
|
+
const context = result.blocks.filter(block => block.block === PromptBlock.Context)
|
|
145
|
+
expect(context).toHaveLength(1)
|
|
146
|
+
expect(context[0]!.text).toContain('## percall')
|
|
147
|
+
expect(context[0]!.text).toContain('first note')
|
|
148
|
+
expect(context[0]!.text).toContain('second note')
|
|
149
|
+
})
|
|
150
|
+
|
|
151
|
+
test('nothing declared composes to no system message at all', async () => {
|
|
152
|
+
const result = await compose(service())
|
|
153
|
+
expect(result.system).toBeNull()
|
|
154
|
+
expect(result.breakpoints).toBe(0)
|
|
155
|
+
})
|
|
156
|
+
})
|
|
157
|
+
|
|
158
|
+
describe('@owlmeans/llm — composed prompt caching', () => {
|
|
159
|
+
test('the system prompt is cached by default, and marks its boundaries', async () => {
|
|
160
|
+
const result = await compose(
|
|
161
|
+
service([skill('a', 'A')], [{ alias: 'pkg', inspect: ctx => ctx.add(PromptBlock.Packages, 'pkg') }]),
|
|
162
|
+
{ role: 'R', skills: ['a'] },
|
|
163
|
+
)
|
|
164
|
+
expect(result.breakpoints).toBe(2)
|
|
165
|
+
expect(Array.isArray(result.system?.content)).toBe(true)
|
|
166
|
+
})
|
|
167
|
+
|
|
168
|
+
test('opting out yields a plain joined string and spends no breakpoints', async () => {
|
|
169
|
+
const result = await compose(service([skill('a', 'A')]), { role: 'R', skills: ['a'], cacheSystem: false })
|
|
170
|
+
expect(result.breakpoints).toBe(0)
|
|
171
|
+
expect(typeof result.system?.content).toBe('string')
|
|
172
|
+
expect(result.system?.content).toContain('## a')
|
|
173
|
+
})
|
|
174
|
+
|
|
175
|
+
test('a provider without explicit markers still gets the blocks, in order', async () => {
|
|
176
|
+
const svc = service([skill('a', 'A')])
|
|
177
|
+
const result = await svc.compose({ role: 'R', skills: ['a'] }, [], { model })
|
|
178
|
+
expect(result.breakpoints).toBe(0)
|
|
179
|
+
expect(result.system?.content).toBe('R\n\n## a\n\nA')
|
|
180
|
+
})
|
|
181
|
+
|
|
182
|
+
// Two stable boundaries at most, so the messages always keep half the request budget.
|
|
183
|
+
test('the system prompt never spends more than half the request budget', async () => {
|
|
184
|
+
const noisy: LlmPromptPlugin = {
|
|
185
|
+
alias: 'noisy',
|
|
186
|
+
compose: ctx => {
|
|
187
|
+
ctx.add(PromptBlock.Packages, 'pkg')
|
|
188
|
+
ctx.add(PromptBlock.Context, 'ctx')
|
|
189
|
+
},
|
|
190
|
+
}
|
|
191
|
+
const result = await compose(service([skill('a', 'A')], [noisy]), { role: 'R', skills: ['a'] })
|
|
192
|
+
expect(result.breakpoints).toBeLessThanOrEqual(2)
|
|
193
|
+
})
|
|
194
|
+
})
|