@owlmeans/llm 0.1.15 → 0.1.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/README.md +3 -4
  2. package/agent-meta/manifest.json +9 -9
  3. package/agent-meta/skills/llm/SKILL.md +54 -4
  4. package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
  5. package/build/consts.d.ts +36 -3
  6. package/build/consts.d.ts.map +1 -1
  7. package/build/consts.js +37 -4
  8. package/build/consts.js.map +1 -1
  9. package/build/execution/service.d.ts.map +1 -1
  10. package/build/execution/service.js +8 -3
  11. package/build/execution/service.js.map +1 -1
  12. package/build/execution/types.d.ts +20 -1
  13. package/build/execution/types.d.ts.map +1 -1
  14. package/build/execution/utils.d.ts +12 -1
  15. package/build/execution/utils.d.ts.map +1 -1
  16. package/build/execution/utils.js +22 -0
  17. package/build/execution/utils.js.map +1 -1
  18. package/build/helpers/cache.d.ts +17 -0
  19. package/build/helpers/cache.d.ts.map +1 -0
  20. package/build/helpers/cache.js +22 -0
  21. package/build/helpers/cache.js.map +1 -0
  22. package/build/helpers/index.d.ts +1 -0
  23. package/build/helpers/index.d.ts.map +1 -1
  24. package/build/helpers/index.js +1 -0
  25. package/build/helpers/index.js.map +1 -1
  26. package/build/helpers/spectate.d.ts.map +1 -1
  27. package/build/helpers/spectate.js +7 -0
  28. package/build/helpers/spectate.js.map +1 -1
  29. package/build/index.d.ts +1 -0
  30. package/build/index.d.ts.map +1 -1
  31. package/build/index.js +1 -0
  32. package/build/index.js.map +1 -1
  33. package/build/model.d.ts +1 -1
  34. package/build/model.d.ts.map +1 -1
  35. package/build/model.js +90 -16
  36. package/build/model.js.map +1 -1
  37. package/build/plugins/anthropic.d.ts.map +1 -1
  38. package/build/plugins/anthropic.js +136 -21
  39. package/build/plugins/anthropic.js.map +1 -1
  40. package/build/plugins/openai.d.ts +9 -0
  41. package/build/plugins/openai.d.ts.map +1 -1
  42. package/build/plugins/openai.js +23 -1
  43. package/build/plugins/openai.js.map +1 -1
  44. package/build/plugins/types.d.ts +38 -5
  45. package/build/plugins/types.d.ts.map +1 -1
  46. package/build/prompt/index.d.ts +5 -0
  47. package/build/prompt/index.d.ts.map +1 -0
  48. package/build/prompt/index.js +4 -0
  49. package/build/prompt/index.js.map +1 -0
  50. package/build/prompt/plugins.d.ts +24 -0
  51. package/build/prompt/plugins.d.ts.map +1 -0
  52. package/build/prompt/plugins.js +64 -0
  53. package/build/prompt/plugins.js.map +1 -0
  54. package/build/prompt/render.d.ts +28 -0
  55. package/build/prompt/render.d.ts.map +1 -0
  56. package/build/prompt/render.js +39 -0
  57. package/build/prompt/render.js.map +1 -0
  58. package/build/prompt/service.d.ts +16 -0
  59. package/build/prompt/service.d.ts.map +1 -0
  60. package/build/prompt/service.js +145 -0
  61. package/build/prompt/service.js.map +1 -0
  62. package/build/prompt/types.d.ts +101 -0
  63. package/build/prompt/types.d.ts.map +1 -0
  64. package/build/prompt/types.js +2 -0
  65. package/build/prompt/types.js.map +1 -0
  66. package/build/service.d.ts.map +1 -1
  67. package/build/service.js +3 -0
  68. package/build/service.js.map +1 -1
  69. package/build/types.d.ts +53 -3
  70. package/build/types.d.ts.map +1 -1
  71. package/build/utils/prompt.d.ts +14 -0
  72. package/build/utils/prompt.d.ts.map +1 -1
  73. package/build/utils/prompt.js +32 -0
  74. package/build/utils/prompt.js.map +1 -1
  75. package/package.json +13 -6
  76. package/src/consts.ts +43 -5
  77. package/src/execution/service.ts +9 -3
  78. package/src/execution/types.ts +21 -2
  79. package/src/execution/utils.ts +28 -1
  80. package/src/helpers/cache.ts +33 -0
  81. package/src/helpers/index.ts +1 -0
  82. package/src/helpers/spectate.ts +10 -0
  83. package/src/index.ts +1 -0
  84. package/src/model.ts +119 -16
  85. package/src/plugins/anthropic.ts +159 -20
  86. package/src/plugins/openai.ts +26 -1
  87. package/src/plugins/types.ts +42 -5
  88. package/src/prompt/index.ts +5 -0
  89. package/src/prompt/plugins.ts +69 -0
  90. package/src/prompt/render.ts +48 -0
  91. package/src/prompt/service.ts +194 -0
  92. package/src/prompt/types.ts +114 -0
  93. package/src/service.ts +3 -0
  94. package/src/types.ts +54 -3
  95. package/src/utils/prompt.ts +33 -0
  96. package/tests/execution.spec.ts +42 -0
  97. package/tests/plugins.spec.ts +279 -14
  98. package/tests/prompt.spec.ts +194 -0
  99. package/agent-meta/instructions/llm.instructions.md +0 -66
@@ -9,7 +9,8 @@ import type {
9
9
  HelperExecution, TaskExecution, WithExecutionService,
10
10
  } from './types.js'
11
11
  import {
12
- composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy, resolveRole,
12
+ composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
13
+ mergePrompt, resolveRole,
13
14
  } from './utils.js'
14
15
 
15
16
  /**
@@ -50,18 +51,21 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
50
51
  level: ExecutionLevel.Project,
51
52
  purpose: { ...input.purpose },
52
53
  policy: { ...input.policy },
54
+ ...(input.prompt != null ? { prompt: { ...input.prompt } } : {}),
53
55
  }) as S['project'],
54
56
 
55
57
  forTask: (parent, input) => {
56
- const { effort, phase, data, ...extras } = input
58
+ const { effort, phase, data, prompt, ...extras } = input
57
59
  const policy = effort != null
58
60
  ? mergePolicy(parent.policy, { effort })
59
61
  : { ...parent.policy }
62
+ const merged = mergePrompt(parent.prompt, prompt)
60
63
 
61
64
  // Spreading the parent carries every collaborator and domain field forward; the
62
65
  // task's own state is composed afterwards, from the seeded resumable fields.
63
66
  const taskExec = {
64
67
  ...parent, ...extras, level: ExecutionLevel.Task, purpose: { ...parent.purpose }, policy,
68
+ ...(merged != null ? { prompt: merged } : {}),
65
69
  } as unknown as TaskExecution
66
70
  ;(taskExec as { state: TaskExecutionState }).state = composeTaskState({
67
71
  ...taskExec,
@@ -78,9 +82,10 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
78
82
  },
79
83
 
80
84
  forHelper: (parent, input) => {
81
- const { role, effort, dedication, ...extras } = input
85
+ const { role, effort, dedication, prompt, ...extras } = input
82
86
  const localPolicy = effort != null ? mergePolicy(parent.policy, { effort }) : parent.policy
83
87
  const scoped = { ...parent, policy: localPolicy } as S['exec']
88
+ const merged = mergePrompt(parent.prompt, prompt)
84
89
 
85
90
  const helperExec = {
86
91
  ...parent,
@@ -90,6 +95,7 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
90
95
  ? { ...parent.purpose, dedication }
91
96
  : { ...parent.purpose },
92
97
  policy: localPolicy,
98
+ ...(merged != null ? { prompt: merged } : {}),
93
99
  role: resolveRole(localPolicy, role),
94
100
  model: self().model(scoped, role),
95
101
  temperatureFactory: self().temperatureFactory(scoped, role),
@@ -1,10 +1,11 @@
1
1
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
2
  import type { InitializedService } from '@owlmeans/context'
3
3
  import type {
4
- ExecutionEffort, ExecutionLevel, ExecutionState, LlmPurpose, ModelConfigOverride,
5
- ModelPolicy, ModelRole, TaskExecutionState,
4
+ ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef, LlmPurpose,
5
+ ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
6
6
  } from '@owlmeans/llm-common'
7
7
  import type { LlmService, TemperatureFactory } from '../types.js'
8
+ import type { PromptService } from '../prompt/types.js'
8
9
 
9
10
  /**
10
11
  * Runtime execution = serializable {@link ExecutionState} + attached collaborators.
@@ -18,6 +19,10 @@ import type { LlmService, TemperatureFactory } from '../types.js'
18
19
  export interface Execution extends ExecutionState {
19
20
  /** Resolver for the model factory — a function so the service can be swapped/cloned. */
20
21
  models: () => LlmService
22
+ /** Resolver for the skill registry / prompt composer. Same late-binding rationale. */
23
+ prompts?: () => PromptService
24
+ /** File access offered to prompt plugins. Declared a collaborator, never snapshotted. */
25
+ files?: FileProviderRef
21
26
  outputErrors?: boolean
22
27
  captureNull?: boolean
23
28
  }
@@ -42,8 +47,12 @@ export interface HelperExecution extends Execution {
42
47
 
43
48
  export interface ProjectExecutionInput {
44
49
  models: () => LlmService
50
+ prompts?: () => PromptService
51
+ files?: FileProviderRef
45
52
  policy: ModelPolicy
46
53
  purpose: LlmPurpose
54
+ /** Baseline role and skills for the whole run. */
55
+ prompt?: PromptPolicy
47
56
  outputErrors?: boolean
48
57
  captureNull?: boolean
49
58
  }
@@ -51,15 +60,23 @@ export interface ProjectExecutionInput {
51
60
  export interface TaskExecutionInput {
52
61
  /** Raise (or lower) the effort tier for this task and everything derived from it. */
53
62
  effort?: ExecutionEffort
63
+ /** Skills (and optionally a role) layered on top of the project's. Skills accumulate. */
64
+ prompt?: PromptPolicy
54
65
  /** Optional seeds for the resumable task state. */
55
66
  phase?: string
56
67
  data?: Record<string, unknown>
57
68
  }
58
69
 
59
70
  export interface HelperExecutionInput {
71
+ /**
72
+ * Which MODEL to use. Distinct from `prompt.role`, which is the system-prompt text
73
+ * defining the persona — one selects hardware, the other writes the job description.
74
+ */
60
75
  role: ModelRole
61
76
  /** Local effort bump without escalating the whole branch. */
62
77
  effort?: ExecutionEffort
78
+ /** The helper's persona and its own skills, layered on top of the task's. */
79
+ prompt?: PromptPolicy
63
80
  /** Refines `purpose.dedication`. */
64
81
  dedication?: string
65
82
  }
@@ -67,6 +84,8 @@ export interface HelperExecutionInput {
67
84
  /** Collaborators re-attached to a state that was restored from storage. */
68
85
  export interface RestoreCollaborators {
69
86
  models?: () => LlmService
87
+ prompts?: () => PromptService
88
+ files?: FileProviderRef
70
89
  }
71
90
 
72
91
  /**
@@ -1,7 +1,7 @@
1
1
  import { ExecutionLevel } from '@owlmeans/llm-common'
2
2
  import type {
3
3
  ExecutionEffort, ExecutionState, ModelConfigOverride, ModelConfigPatch,
4
- ModelPolicy, ModelRole, TaskExecutionState,
4
+ ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
5
5
  } from '@owlmeans/llm-common'
6
6
  import { EFFORT_TABLE } from '../consts.js'
7
7
  import type { Execution, TaskExecution } from './types.js'
@@ -19,6 +19,33 @@ export const mergePolicy = (base: ModelPolicy, patch: Partial<ModelPolicy>): Mod
19
19
  : undefined,
20
20
  })
21
21
 
22
+ /**
23
+ * Overlay a prompt policy onto the one inherited from the parent level.
24
+ *
25
+ * Skills ACCUMULATE — a task adds to what the project declared, a helper adds to the
26
+ * task — because that is how a capability set is built up as work narrows. The role is
27
+ * replaced instead: the deepest level that names one owns the persona.
28
+ *
29
+ * The union preserves first-seen order and de-duplicates, so the composed prompt is
30
+ * byte-identical no matter how many levels contributed the same skill.
31
+ */
32
+ export const mergePrompt = (
33
+ base: PromptPolicy | undefined,
34
+ patch: PromptPolicy | undefined,
35
+ ): PromptPolicy | undefined => {
36
+ if (base == null && patch == null) {
37
+ return undefined
38
+ }
39
+ const skills = [...new Set([...(base?.skills ?? []), ...(patch?.skills ?? [])])]
40
+
41
+ return {
42
+ ...base,
43
+ ...patch,
44
+ ...(base?.role != null || patch?.role != null ? { role: patch?.role ?? base?.role } : {}),
45
+ ...(skills.length > 0 ? { skills } : {}),
46
+ }
47
+ }
48
+
22
49
  /** Apply the policy's role→role remap. */
23
50
  export const resolveRole = (policy: ModelPolicy, role: ModelRole): ModelRole =>
24
51
  (policy.roleOverrides?.[role] as ModelRole | undefined) ?? role
@@ -0,0 +1,33 @@
1
+ import type { UsageMetadata } from '@langchain/core/messages'
2
+ import type { CacheUsage } from '@owlmeans/llm-common'
3
+
4
+ /** LangChain normalizes every provider's cache accounting into these two fields. */
5
+ interface InputTokenDetails {
6
+ cache_read?: number
7
+ cache_creation?: number
8
+ }
9
+
10
+ /**
11
+ * Prompt-cache accounting for one completion.
12
+ *
13
+ * This is the only honest answer to "is caching actually working". A composed prefix can
14
+ * look perfectly stable and still miss on every call — a stray timestamp, a set iterated
15
+ * in a different order, a tool list rebuilt per request. If `read` stays at zero across
16
+ * repeated calls that share a prefix, something is invalidating it; diff the rendered
17
+ * blocks between two calls to find out what.
18
+ */
19
+ export const readCacheUsage = (message: { usage_metadata?: UsageMetadata }): CacheUsage => {
20
+ const usage = message.usage_metadata
21
+ const details = usage?.input_token_details as InputTokenDetails | undefined
22
+
23
+ return {
24
+ read: details?.cache_read ?? 0,
25
+ creation: details?.cache_creation ?? 0,
26
+ input: usage?.input_tokens ?? 0,
27
+ output: usage?.output_tokens ?? 0,
28
+ }
29
+ }
30
+
31
+ /** `true` when the provider reported any cache activity at all. */
32
+ export const hasCacheActivity = (usage: CacheUsage): boolean =>
33
+ usage.read > 0 || usage.creation > 0
@@ -3,3 +3,4 @@ export * from './retry.js'
3
3
  export * from './json.js'
4
4
  export * from './messages.js'
5
5
  export * from './spectate.js'
6
+ export * from './cache.js'
@@ -3,6 +3,7 @@ import type { UsageMetadata } from '@langchain/core/messages'
3
3
  import { SpectatorContentType } from '@owlmeans/llm-common'
4
4
  import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
5
5
  import type { LlmSpectator, ModelInputItem } from '../types.js'
6
+ import { hasCacheActivity, readCacheUsage } from './cache.js'
6
7
 
7
8
  /** Normalize one prompt message into the spectator's storage shape. */
8
9
  const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
@@ -63,5 +64,14 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
63
64
  completion.content = message.content as unknown as string
64
65
  }
65
66
 
67
+ // Silent unless the provider reported cache activity, so it costs nothing when
68
+ // caching is off — and is the one signal that tells a stable prefix from a broken one.
69
+ const cache = readCacheUsage(message)
70
+ if (hasCacheActivity(cache)) {
71
+ console.log(
72
+ `Prompt cache [${action}]: read ${cache.read}, written ${cache.creation}, uncached ${cache.input}`
73
+ )
74
+ }
75
+
66
76
  return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
67
77
  }
package/src/index.ts CHANGED
@@ -6,6 +6,7 @@ export * from './model.js'
6
6
  export * from './service.js'
7
7
  export * from './helpers/index.js'
8
8
  export * from './execution/index.js'
9
+ export * from './prompt/index.js'
9
10
  export type * from './plugins/types.js'
10
11
  export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
11
12
  export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
package/src/model.ts CHANGED
@@ -1,12 +1,12 @@
1
1
  import { Ajv } from 'ajv'
2
2
  import type { JSONSchemaType } from 'ajv'
3
- import { AIMessage } from '@langchain/core/messages'
4
- import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
3
+ import { AIMessage, BaseMessage } from '@langchain/core/messages'
4
+ import type { AIMessageChunk, MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
5
5
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
6
6
  import { StructuredMode } from '@owlmeans/llm-common'
7
7
  import type { NullKind } from '@owlmeans/llm-common'
8
8
  import {
9
- DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
9
+ DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
10
10
  } from './consts.js'
11
11
  import { LlmModelError } from './errors.js'
12
12
  import { pluginFor, pluginOf } from './plugins/index.js'
@@ -18,7 +18,7 @@ import { spectate } from './helpers/spectate.js'
18
18
  import { idleTimeout, readConfig } from './utils/config.js'
19
19
  import { reportNull } from './utils/null-report.js'
20
20
  import type { NullReportParams } from './utils/null-report.js'
21
- import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
21
+ import { applyNoThink, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
22
22
  import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
23
23
  import { streamWithDeadline } from './utils/stream.js'
24
24
  import type {
@@ -28,6 +28,48 @@ import type {
28
28
 
29
29
  type StreamOptions = Parameters<BaseChatModel['stream']>[1]
30
30
 
31
+ const isSystem = (msg: MessageFieldWithRole): boolean =>
32
+ msg instanceof BaseMessage ? msg.getType() === 'system' : `${msg.role}` === 'system'
33
+
34
+ const textOf = (content: MessageContent | undefined): string => {
35
+ if (typeof content === 'string') {
36
+ return content.trim()
37
+ }
38
+ if (Array.isArray(content)) {
39
+ return content
40
+ .map(part => {
41
+ const text = (part as unknown as { text?: unknown }).text
42
+ return typeof text === 'string' ? text : ''
43
+ })
44
+ .filter(text => text !== '')
45
+ .join('\n\n')
46
+ .trim()
47
+ }
48
+ return ''
49
+ }
50
+
51
+ /**
52
+ * Detach the caller's LEADING system messages and return their text.
53
+ *
54
+ * They are re-emitted as the `Context` block of the composed prompt, which is what keeps
55
+ * a caller that still builds its own `SystemMessage` working unchanged — the text simply
56
+ * travels a different route and lands in the same place. Only the leading run is taken:
57
+ * a system message deliberately placed mid-conversation is an operator instruction whose
58
+ * position carries meaning, and moving it would change what the model sees.
59
+ */
60
+ const takeLeadingSystem = (msgs: MessageFieldWithRole[]): string[] => {
61
+ const carried: string[] = []
62
+ while (msgs.length > 0 && isSystem(msgs[0])) {
63
+ const [msg] = msgs.splice(0, 1)
64
+ const text = textOf(msg.content)
65
+ if (text !== '') {
66
+ carried.push(text)
67
+ }
68
+ }
69
+
70
+ return carried
71
+ }
72
+
31
73
  /**
32
74
  * Build the four-method model API on top of a LangChain chat model.
33
75
  *
@@ -43,6 +85,9 @@ export const makeLlmModel = ({
43
85
  captureNull = false,
44
86
  retries = DEFAULT_MODEL_RETRIES,
45
87
  purpose,
88
+ prompt,
89
+ prompts,
90
+ files,
46
91
  }: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
47
92
 
48
93
  const ajv = new Ajv({ strict: false })
@@ -53,15 +98,67 @@ export const makeLlmModel = ({
53
98
  const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
54
99
  const timeout = idleTimeout(config)
55
100
 
56
- /** Normalize, then apply every in-place prompt adaptation, in dependency order. */
57
- const prepare = (input: ModelInput, useCache: boolean, cacheMax: number, json: boolean): MessageFieldWithRole[] => {
101
+ /**
102
+ * Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
103
+ * dependency order.
104
+ *
105
+ * With no prompt service wired this is exactly what it always was — the caller's
106
+ * messages, a JSON nudge, `/no_think`, cache markers. With one, the caller's leading
107
+ * system text is folded into a composed prompt whose stable sections come first, which
108
+ * is the whole point: a prompt cache is a PREFIX match, so the bytes every call shares
109
+ * have to be physically ahead of the bytes that differ.
110
+ */
111
+ const prepare = async (
112
+ input: ModelInput,
113
+ action: string,
114
+ useCache: boolean,
115
+ cacheMax: number,
116
+ json: boolean,
117
+ callSkills?: string[],
118
+ ): Promise<MessageFieldWithRole[]> => {
58
119
  const msgs = normalizeInput(input)
120
+ // The caller may hand back messages this pipeline marked on a PREVIOUS call — the
121
+ // markers live on its own objects. The budget is per request, so clear them and
122
+ // re-place our own below; otherwise they accumulate until the provider 400s.
123
+ stripCacheMarkers(msgs)
124
+ let reserved = 0
125
+
126
+ if (prompts != null) {
127
+ const carried = takeLeadingSystem(msgs)
128
+ const composed = await prompts().compose(
129
+ {
130
+ ...prompt,
131
+ context: [...(prompt?.context ?? []), ...carried],
132
+ callSkills: callSkills ?? prompt?.callSkills,
133
+ },
134
+ msgs,
135
+ { model, provider: plugin, purpose, action, cacheMax, files },
136
+ )
137
+ if (composed.system != null) {
138
+ msgs.unshift(composed.system)
139
+ reserved = composed.breakpoints
140
+ } else {
141
+ // Defensive: nothing was contributed, so hand the caller's own text straight back.
142
+ for (let i = carried.length - 1; i >= 0; i--) {
143
+ msgs.unshift({ role: 'system', content: carried[i] })
144
+ }
145
+ }
146
+ }
147
+
59
148
  if (json) ensureJsonMention(msgs)
60
149
  applyNoThink(msgs, config.disableThinking)
61
150
  // Cache markers replace string content with content blocks, so they must go last.
62
- if (plugin?.patchCache?.(msgs, { model, useCache, cacheMax }) === true) {
63
- console.log(`Prompt caching enabled for ${plugin.type} (up to ${cacheMax} breakpoints)`)
151
+ const ttl = prompt?.cacheTtl
152
+ const marked = plugin?.patchCache?.(msgs, {
153
+ model, useCache, cacheMax, reserved, ...(ttl != null ? { ttl } : {}),
154
+ })
155
+ if (reserved > 0 || marked === true) {
156
+ console.log(
157
+ `Prompt caching for ${plugin?.type}: ${reserved} system breakpoint(s)`
158
+ + `${marked === true ? ', 1 message breakpoint' : ''}`
159
+ )
64
160
  }
161
+
65
162
  return msgs
66
163
  }
67
164
 
@@ -197,8 +294,11 @@ export const makeLlmModel = ({
197
294
  }
198
295
 
199
296
  const helper: LlmModel = {
200
- ask: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmAskOptions) => {
201
- const msgs = prepare(input, useCache, cacheMax, false)
297
+ ask: async (
298
+ input,
299
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmAskOptions
300
+ ) => {
301
+ const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
202
302
  return withRetry({ retries, outputErrors }, async i => {
203
303
  const refined = refineModel(i)
204
304
  console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
@@ -242,8 +342,11 @@ export const makeLlmModel = ({
242
342
  })
243
343
  },
244
344
 
245
- talk: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmTalkOptions) => {
246
- const msgs = prepare(input, useCache, cacheMax, false)
345
+ talk: async (
346
+ input,
347
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmTalkOptions
348
+ ) => {
349
+ const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
247
350
  return withRetry({ retries, outputErrors }, async i => {
248
351
  const refined = refineModel(i)
249
352
  console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
@@ -277,9 +380,9 @@ export const makeLlmModel = ({
277
380
  invoke: async <T>(
278
381
  input: ModelInput,
279
382
  schema: JSONSchemaType<T>,
280
- { temperature, ref, filter, action, useCache = false, cacheMax = 4 }: LlmInvokeOptions<T>
383
+ { temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmInvokeOptions<T>
281
384
  ) => {
282
- const msgs = prepare(input, useCache, cacheMax, true)
385
+ const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
283
386
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
284
387
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
285
388
 
@@ -325,9 +428,9 @@ export const makeLlmModel = ({
325
428
  request: async <T>(
326
429
  input: ModelInput,
327
430
  schema: JSONSchemaType<T>,
328
- { ref, filter, action, useCache = false, cacheMax = 4 }: LlmRequestOptions
431
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmRequestOptions
329
432
  ) => {
330
- const msgs = prepare(input, useCache, cacheMax, true)
433
+ const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
331
434
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
332
435
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
333
436
 
@@ -1,9 +1,12 @@
1
1
  import { ChatAnthropic } from '@langchain/anthropic'
2
2
  import { BadRequestError } from '@anthropic-ai/sdk'
3
3
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
4
- import { ModelProvider, StructuredMode } from '@owlmeans/llm-common'
4
+ import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
5
+ import { ModelProvider, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
6
+ import type { CacheTtl } from '@owlmeans/llm-common'
5
7
  import type { LlmPlugin } from './types.js'
6
- import { MAX_CACHE_BREAKPOINTS } from '../consts.js'
8
+ import { CHARS_PER_TOKEN, MAX_CACHE_BREAKPOINTS, MIN_CACHEABLE_TOKENS } from '../consts.js'
9
+ import { readConfig } from '../utils/config.js'
7
10
  import { escalateMaxTokens, makeClientOptions } from './utils.js'
8
11
 
9
12
  /** Model-name prefix that supports prompt caching through `cache_control` markers. */
@@ -11,6 +14,64 @@ const CACHEABLE_PREFIX = 'claude-'
11
14
 
12
15
  export const ANTHROPIC_FAMILY = 'anthropic'
13
16
 
17
+ type ContentBlock = Record<string, unknown>
18
+
19
+ const supportsCache = (model: BaseChatModel): boolean =>
20
+ (model as ChatAnthropic).modelName?.startsWith(CACHEABLE_PREFIX) === true
21
+
22
+ /**
23
+ * Shortest prefix worth a breakpoint, in characters. Anthropic silently declines to
24
+ * create an entry below its own per-model minimum, so a marker there wastes one of the
25
+ * four breakpoints and reports a cache that was never written.
26
+ */
27
+ const minCacheableChars = (model: BaseChatModel): number =>
28
+ (readConfig(model).cacheMinTokens ?? MIN_CACHEABLE_TOKENS) * CHARS_PER_TOKEN
29
+
30
+ /**
31
+ * The marker itself. `ttl` is omitted for the 5-minute default so the emitted bytes stay
32
+ * the classic shape — a request that differs only in an explicit `"ttl": "5m"` would not
33
+ * match a prefix cached without it.
34
+ */
35
+ const marker = (ttl: CacheTtl): ContentBlock =>
36
+ ttl === '1h' ? { type: 'ephemeral', ttl: '1h' } : { type: 'ephemeral' }
37
+
38
+ const contentLength = (content: MessageContent | undefined): number => {
39
+ if (typeof content === 'string') {
40
+ return content.length
41
+ }
42
+ if (Array.isArray(content)) {
43
+ return content.reduce<number>((sum, part) => {
44
+ const text = (part as unknown as { text?: unknown }).text
45
+ return sum + (typeof text === 'string' ? text.length : 0)
46
+ }, 0)
47
+ }
48
+ return 0
49
+ }
50
+
51
+ /**
52
+ * Put a breakpoint on a message's LAST content block, lifting string content into a block
53
+ * so the marker has somewhere to live. Idempotent, and it never mutates a block the
54
+ * caller owns — the array is rebuilt around a fresh copy of the final entry.
55
+ */
56
+ const markMessage = (msg: MessageFieldWithRole, mark: ContentBlock): boolean => {
57
+ if (typeof msg.content === 'string') {
58
+ msg.content = [{ type: 'text', text: msg.content, cache_control: mark }] as unknown as MessageContent
59
+ return true
60
+ }
61
+ if (Array.isArray(msg.content) && msg.content.length > 0) {
62
+ const blocks = [...msg.content] as ContentBlock[]
63
+ const last = blocks[blocks.length - 1]
64
+ if (last.cache_control != null) {
65
+ return true
66
+ }
67
+ blocks[blocks.length - 1] = { ...last, cache_control: mark }
68
+ msg.content = blocks as unknown as MessageContent
69
+ return true
70
+ }
71
+
72
+ return false
73
+ }
74
+
14
75
  export const anthropicPlugin: LlmPlugin = {
15
76
  type: ModelProvider.Anthropic,
16
77
 
@@ -70,28 +131,106 @@ export const anthropicPlugin: LlmPlugin = {
70
131
  },
71
132
 
72
133
  /**
73
- * Mark the leading messages with an ephemeral `cache_control` breakpoint. Anthropic
74
- * allows at most {@link MAX_CACHE_BREAKPOINTS}; string content is lifted into a
75
- * single text block so the marker has somewhere to live.
134
+ * Render the composed system prompt as Anthropic content blocks, one per section, with
135
+ * a breakpoint on each stability boundary:
136
+ *
137
+ * - after `Role` + `Skills` — the region every call of this role shares;
138
+ * - after `Packages` — varies with what the request mentions, so it gets its own
139
+ * entry and can never invalidate the block above it;
140
+ * - after the last block — so the whole system prompt is cached, which is the
141
+ * default this layer promises.
142
+ *
143
+ * Boundaries that coincide collapse into one. Marking stops as soon as the budget is
144
+ * spent, earliest boundary first — the earliest prefix is the one most calls share.
76
145
  */
77
- patchCache: (msgs, { model, useCache, cacheMax }) => {
78
- if (!useCache || !(model as ChatAnthropic).modelName?.startsWith(CACHEABLE_PREFIX)) return false
79
- const max = Math.min(cacheMax, MAX_CACHE_BREAKPOINTS)
80
- let i = 0
81
- for (const msg of msgs) {
82
- msg.content = typeof msg.content === 'string' ? [{
83
- type: 'text',
84
- text: msg.content,
85
- cache_control: { type: 'ephemeral' },
86
- }] : msg.content
87
- if (++i > max - 1) break
146
+ patchSystem: (blocks, { model, cacheMax, ttl }) => {
147
+ const content: ContentBlock[] = blocks.map(block => ({ type: 'text', text: block.text }))
148
+ if (!supportsCache(model) || cacheMax < 1) {
149
+ return { content: content as unknown as MessageContent, breakpoints: 0 }
88
150
  }
89
- return true
151
+
152
+ const lastOf = (...wanted: PromptBlock[]): number =>
153
+ blocks.reduce((found, block, i) => wanted.includes(block.block) ? i : found, -1)
154
+
155
+ // Closing the prompt is worth a breakpoint only when the last block is STABLE. A
156
+ // trailing `Context` changes every call, so marking it would pay a cache write every
157
+ // time and never read one back — it burns a breakpoint to buy nothing. A prompt that
158
+ // is ONLY context (a caller that has not adopted role/skills) is still worth marking,
159
+ // because there it IS the stable part.
160
+ const last = blocks.length - 1
161
+ const closing = blocks[last].block === PromptBlock.Context && blocks.length > 1 ? -1 : last
162
+
163
+ const boundaries = [...new Set([
164
+ lastOf(PromptBlock.Role, PromptBlock.Skills),
165
+ lastOf(PromptBlock.Packages),
166
+ closing,
167
+ ].filter(index => index >= 0))].sort((a, b) => a - b)
168
+
169
+ const minChars = minCacheableChars(model)
170
+ let consumed = 0
171
+ let chars = 0
172
+ let next = 0
173
+ for (let i = 0; i < content.length; i++) {
174
+ chars += blocks[i].text.length
175
+ if (i !== boundaries[next]) {
176
+ continue
177
+ }
178
+ next++
179
+ if (chars < minChars || consumed >= cacheMax) {
180
+ continue
181
+ }
182
+ content[i].cache_control = marker(ttl)
183
+ consumed++
184
+ }
185
+
186
+ return { content: content as unknown as MessageContent, breakpoints: consumed }
187
+ },
188
+
189
+ /**
190
+ * One breakpoint, at the end of the stable message prefix (`cacheMax` messages).
191
+ *
192
+ * Not one marker per message: the request budget is {@link MAX_CACHE_BREAKPOINTS} in
193
+ * total across tools, system and messages, and the system prompt — the part that is
194
+ * genuinely identical between calls — has first claim on it. `reserved` is what the
195
+ * system prompt already spent.
196
+ */
197
+ patchCache: (msgs, { model, useCache, cacheMax, reserved = 0, ttl = '5m' }) => {
198
+ if (!useCache || !supportsCache(model) || msgs.length === 0) {
199
+ return false
200
+ }
201
+ if (Math.min(cacheMax, MAX_CACHE_BREAKPOINTS - reserved) < 1) {
202
+ return false
203
+ }
204
+
205
+ // The final message is the per-call payload, and `ensureJsonMention` / `applyNoThink`
206
+ // append to it — including it in the prefix would write a fresh entry every call and
207
+ // read none. The stable prefix therefore stops one short of the end.
208
+ const index = Math.min(cacheMax, msgs.length - 1) - 1
209
+ const target = index >= 0 ? msgs[index] : null
210
+ if (target == null) {
211
+ return false
212
+ }
213
+ const chars = msgs.slice(0, index + 1)
214
+ .reduce((sum, msg) => sum + contentLength(msg.content), 0)
215
+ if (chars < minCacheableChars(model)) {
216
+ return false
217
+ }
218
+
219
+ return markMessage(target, marker(ttl))
90
220
  },
91
221
 
92
222
  /**
93
- * A malformed request (bad schema, unsupported parameter, oversized `max_tokens`)
94
- * cannot be fixed by retrying — surface it immediately instead of burning the budget.
223
+ * A malformed request (bad schema, unsupported parameter, oversized `max_tokens`, too
224
+ * many cache breakpoints) cannot be fixed by retrying — surface it immediately instead
225
+ * of burning the budget.
226
+ *
227
+ * The `status` check is not redundant with the `instanceof`: `@langchain/anthropic`
228
+ * carries its OWN nested copy of `@anthropic-ai/sdk`, so the error it throws is an
229
+ * instance of a DIFFERENT `BadRequestError` class than the one imported here and the
230
+ * `instanceof` silently fails. That turned every fatal 400 into eight full retries —
231
+ * a single malformed request became minutes of thrash with the real cause buried.
95
232
  */
96
- isFatal: e => e instanceof BadRequestError ? e : null,
233
+ isFatal: e => e instanceof BadRequestError || (e as { status?: unknown })?.status === 400
234
+ ? e as Error
235
+ : null,
97
236
  }