@owlmeans/llm 0.1.15 → 0.1.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/README.md +3 -4
  2. package/agent-meta/manifest.json +9 -9
  3. package/agent-meta/skills/llm/SKILL.md +54 -4
  4. package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
  5. package/build/consts.d.ts +36 -3
  6. package/build/consts.d.ts.map +1 -1
  7. package/build/consts.js +37 -4
  8. package/build/consts.js.map +1 -1
  9. package/build/execution/service.d.ts.map +1 -1
  10. package/build/execution/service.js +8 -3
  11. package/build/execution/service.js.map +1 -1
  12. package/build/execution/types.d.ts +20 -1
  13. package/build/execution/types.d.ts.map +1 -1
  14. package/build/execution/utils.d.ts +12 -1
  15. package/build/execution/utils.d.ts.map +1 -1
  16. package/build/execution/utils.js +22 -0
  17. package/build/execution/utils.js.map +1 -1
  18. package/build/helpers/cache.d.ts +17 -0
  19. package/build/helpers/cache.d.ts.map +1 -0
  20. package/build/helpers/cache.js +22 -0
  21. package/build/helpers/cache.js.map +1 -0
  22. package/build/helpers/index.d.ts +1 -0
  23. package/build/helpers/index.d.ts.map +1 -1
  24. package/build/helpers/index.js +1 -0
  25. package/build/helpers/index.js.map +1 -1
  26. package/build/helpers/spectate.d.ts.map +1 -1
  27. package/build/helpers/spectate.js +7 -0
  28. package/build/helpers/spectate.js.map +1 -1
  29. package/build/index.d.ts +1 -0
  30. package/build/index.d.ts.map +1 -1
  31. package/build/index.js +1 -0
  32. package/build/index.js.map +1 -1
  33. package/build/model.d.ts +1 -1
  34. package/build/model.d.ts.map +1 -1
  35. package/build/model.js +90 -16
  36. package/build/model.js.map +1 -1
  37. package/build/plugins/anthropic.d.ts.map +1 -1
  38. package/build/plugins/anthropic.js +136 -21
  39. package/build/plugins/anthropic.js.map +1 -1
  40. package/build/plugins/openai.d.ts +9 -0
  41. package/build/plugins/openai.d.ts.map +1 -1
  42. package/build/plugins/openai.js +23 -1
  43. package/build/plugins/openai.js.map +1 -1
  44. package/build/plugins/types.d.ts +38 -5
  45. package/build/plugins/types.d.ts.map +1 -1
  46. package/build/prompt/index.d.ts +5 -0
  47. package/build/prompt/index.d.ts.map +1 -0
  48. package/build/prompt/index.js +4 -0
  49. package/build/prompt/index.js.map +1 -0
  50. package/build/prompt/plugins.d.ts +24 -0
  51. package/build/prompt/plugins.d.ts.map +1 -0
  52. package/build/prompt/plugins.js +64 -0
  53. package/build/prompt/plugins.js.map +1 -0
  54. package/build/prompt/render.d.ts +28 -0
  55. package/build/prompt/render.d.ts.map +1 -0
  56. package/build/prompt/render.js +39 -0
  57. package/build/prompt/render.js.map +1 -0
  58. package/build/prompt/service.d.ts +16 -0
  59. package/build/prompt/service.d.ts.map +1 -0
  60. package/build/prompt/service.js +145 -0
  61. package/build/prompt/service.js.map +1 -0
  62. package/build/prompt/types.d.ts +101 -0
  63. package/build/prompt/types.d.ts.map +1 -0
  64. package/build/prompt/types.js +2 -0
  65. package/build/prompt/types.js.map +1 -0
  66. package/build/service.d.ts.map +1 -1
  67. package/build/service.js +3 -0
  68. package/build/service.js.map +1 -1
  69. package/build/types.d.ts +53 -3
  70. package/build/types.d.ts.map +1 -1
  71. package/build/utils/prompt.d.ts +14 -0
  72. package/build/utils/prompt.d.ts.map +1 -1
  73. package/build/utils/prompt.js +32 -0
  74. package/build/utils/prompt.js.map +1 -1
  75. package/package.json +13 -6
  76. package/src/consts.ts +43 -5
  77. package/src/execution/service.ts +9 -3
  78. package/src/execution/types.ts +21 -2
  79. package/src/execution/utils.ts +28 -1
  80. package/src/helpers/cache.ts +33 -0
  81. package/src/helpers/index.ts +1 -0
  82. package/src/helpers/spectate.ts +10 -0
  83. package/src/index.ts +1 -0
  84. package/src/model.ts +119 -16
  85. package/src/plugins/anthropic.ts +159 -20
  86. package/src/plugins/openai.ts +26 -1
  87. package/src/plugins/types.ts +42 -5
  88. package/src/prompt/index.ts +5 -0
  89. package/src/prompt/plugins.ts +69 -0
  90. package/src/prompt/render.ts +48 -0
  91. package/src/prompt/service.ts +194 -0
  92. package/src/prompt/types.ts +114 -0
  93. package/src/service.ts +3 -0
  94. package/src/types.ts +54 -3
  95. package/src/utils/prompt.ts +33 -0
  96. package/tests/execution.spec.ts +42 -0
  97. package/tests/plugins.spec.ts +279 -14
  98. package/tests/prompt.spec.ts +194 -0
  99. package/agent-meta/instructions/llm.instructions.md +0 -66
@@ -33,6 +33,17 @@ export const openAiFamily = {
33
33
  json_schema: { name: toolName, schema, strict: false },
34
34
  }),
35
35
 
36
+ /**
37
+ * A 400 means the request itself is malformed — a schema the endpoint rejects, an
38
+ * unsupported parameter, a `max_tokens` above the model's per-request limit. Retrying
39
+ * re-sends the same shape (and the escalator raises `max_tokens`, making the last case
40
+ * strictly worse), so eight attempts only bury the real message. Matched on `status`
41
+ * rather than an SDK class: aggregators and nested SDK copies throw their own error
42
+ * types, and an `instanceof` against one of them silently never matches.
43
+ */
44
+ isFatal: (e: unknown): Error | null =>
45
+ (e as { status?: unknown })?.status === 400 ? e as Error : null,
46
+
36
47
  refine: ({ base, attempt, temperature, maxOutputCap }: LlmRefineParams): BaseChatModel => {
37
48
  const model = base as ChatOpenAI
38
49
  const currentTemperature = temperature ?? model.temperature ?? 0
@@ -75,10 +86,22 @@ export const openAiPlugin: LlmPlugin = {
75
86
  structuredMode: (config: ModelConfig): StructuredMode =>
76
87
  config.structuredOutput === false ? StructuredMode.Tool : StructuredMode.Native,
77
88
 
78
- build: ({ config, secret, callbacks }) => {
89
+ build: ({ alias, config, secret, callbacks }) => {
79
90
  const model = config.model ??= 'gpt-5.4-mini'
80
91
  const configuration = makeConfiguration({ baseURL: undefined, headers: config.headers })
81
92
 
93
+ // OpenAI's prompt cache is automatic and prefix-based — there is nothing to mark. The
94
+ // one lever a client has is routing: requests are dispatched by a hash of the prompt's
95
+ // opening tokens, and `prompt_cache_key` is mixed into that hash, so requests sharing
96
+ // a key land on the same backend and can actually hit each other's entries. The key
97
+ // must be stable and low-cardinality; the config alias IS the role, which is exactly
98
+ // the granularity at which a system prefix is shared.
99
+ //
100
+ // Deliberately NOT done for the `compatible` plugin: aggregators there run with
101
+ // `provider.require_parameters`, and an unknown top-level field can exclude every
102
+ // serving provider from the route.
103
+ const modelKwargs = { prompt_cache_key: config.cacheKey ?? alias }
104
+
82
105
  // The Responses API models reject `temperature`/`topP`.
83
106
  if (RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))) {
84
107
  return new ChatOpenAI({
@@ -89,6 +112,7 @@ export const openAiPlugin: LlmPlugin = {
89
112
  useResponsesApi: true,
90
113
  metadata: { config },
91
114
  callbacks,
115
+ modelKwargs,
92
116
  ...configuration,
93
117
  })
94
118
  }
@@ -102,6 +126,7 @@ export const openAiPlugin: LlmPlugin = {
102
126
  maxRetries: 5,
103
127
  metadata: { config },
104
128
  callbacks,
129
+ modelKwargs,
105
130
  ...configuration,
106
131
  })
107
132
  },
@@ -1,7 +1,7 @@
1
1
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
2
  import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/core/callbacks/base'
3
- import type { MessageFieldWithRole } from '@langchain/core/messages'
4
- import type { StructuredMode } from '@owlmeans/llm-common'
3
+ import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
4
+ import type { CacheTtl, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
5
5
  import type { ModelConfig } from '../types.js'
6
6
 
7
7
  export interface LlmBuildParams {
@@ -32,7 +32,32 @@ export interface LlmCacheParams {
32
32
  /** The ORIGINAL (unrefined) model — refined instances do not always keep the name. */
33
33
  model: BaseChatModel
34
34
  useCache: boolean
35
+ /** How many leading messages form the stable prefix worth caching. */
35
36
  cacheMax: number
37
+ /** Breakpoints the composed system prompt already spent on this request. */
38
+ reserved?: number
39
+ ttl?: CacheTtl
40
+ }
41
+
42
+ /** One rendered section of the composed system prompt, in emission order. */
43
+ export interface LlmSystemBlock {
44
+ block: PromptBlock
45
+ text: string
46
+ }
47
+
48
+ export interface LlmSystemCacheParams {
49
+ /** The ORIGINAL (unrefined) model. */
50
+ model: BaseChatModel
51
+ /** Breakpoint budget the system prompt may spend. */
52
+ cacheMax: number
53
+ ttl: CacheTtl
54
+ }
55
+
56
+ /** Provider-native rendering of the composed system prompt. */
57
+ export interface LlmSystemRender {
58
+ content: MessageContent
59
+ /** Breakpoints consumed — subtracted from the message-level budget. */
60
+ breakpoints: number
36
61
  }
37
62
 
38
63
  /**
@@ -78,12 +103,24 @@ export interface LlmPlugin {
78
103
  responseFormat?: (toolName: string, schema: unknown) => Record<string, unknown>
79
104
 
80
105
  /**
81
- * Mark leading messages as cacheable in-place. Returns `true` when caching was
82
- * actually applied (the model may not support it). Omit for providers with no
83
- * explicit prompt-cache markers.
106
+ * Mark the stable message prefix as cacheable, in-place. ONE breakpoint at the end of
107
+ * that prefix — not one per message: the budget is four per request and the system
108
+ * prompt has first claim on it. Returns `true` when a marker was actually placed.
109
+ * Omit for providers with no explicit prompt-cache markers.
84
110
  */
85
111
  patchCache?: (msgs: MessageFieldWithRole[], params: LlmCacheParams) => boolean
86
112
 
113
+ /**
114
+ * Render the composed system blocks into provider-native content, placing cache
115
+ * breakpoints at the stability boundaries between blocks (see `PromptBlock`).
116
+ *
117
+ * Return `null` — or omit the method entirely — for providers whose prompt cache is
118
+ * automatic and prefix-based (OpenAI and friends): the service then joins the blocks
119
+ * into a plain string, which is all those providers need, since the block ORDER is
120
+ * what makes their prefix stable.
121
+ */
122
+ patchSystem?: (blocks: LlmSystemBlock[], params: LlmSystemCacheParams) => LlmSystemRender | null
123
+
87
124
  /**
88
125
  * Classify a thrown error as fatal for the retry loop. Return the error to throw
89
126
  * immediately, or `null` to let it be retried.
@@ -0,0 +1,5 @@
1
+
2
+ export type * from './types.js'
3
+ export * from './render.js'
4
+ export * from './plugins.js'
5
+ export * from './service.js'
@@ -0,0 +1,69 @@
1
+ import { PromptBlock } from '@owlmeans/llm-common'
2
+ import type { SkillDefinition } from '@owlmeans/llm-common'
3
+ import { joinChunks, renderSkill, sortSkills } from './render.js'
4
+ import type { LlmPromptPlugin } from './types.js'
5
+
6
+ /**
7
+ * Block 0 — the base system prompt that tells the model who it is.
8
+ *
9
+ * The most stable thing in the whole request, so it goes first and every cache boundary
10
+ * sits behind it.
11
+ */
12
+ export const rolePlugin: LlmPromptPlugin = {
13
+ alias: 'role',
14
+ order: 0,
15
+ compose: ctx => {
16
+ if (ctx.input.role != null && ctx.input.role.trim() !== '') {
17
+ ctx.add(PromptBlock.Role, ctx.input.role)
18
+ }
19
+ },
20
+ }
21
+
22
+ /**
23
+ * Block 1 — the declared capabilities, rendered in a deterministic order.
24
+ *
25
+ * Registry skills and inline skills are merged by alias with inline winning, so a caller
26
+ * can override one registered entry without forking the catalogue.
27
+ */
28
+ export const skillsPlugin: LlmPromptPlugin = {
29
+ alias: 'skills',
30
+ order: 10,
31
+ compose: ctx => {
32
+ const declared = ctx.resolve(ctx.input.skills ?? [])
33
+ const merged = new Map<string, SkillDefinition>()
34
+ for (const skill of [...declared, ...(ctx.input.inline ?? [])]) {
35
+ merged.set(skill.alias, skill)
36
+ }
37
+ for (const skill of sortSkills([...merged.values()])) {
38
+ ctx.add(skill.block ?? PromptBlock.Skills, renderSkill(skill))
39
+ }
40
+ },
41
+ }
42
+
43
+ /**
44
+ * Block 3 — whatever the caller handed over verbatim, plus any skills requested for this
45
+ * one call. Emitted last and never marked cacheable: its content varies per request by
46
+ * definition, and a varying tail must not sit inside a prefix other calls depend on.
47
+ */
48
+ export const contextPlugin: LlmPromptPlugin = {
49
+ alias: 'context',
50
+ order: 90,
51
+ compose: ctx => {
52
+ // Everything volatile is merged into ONE chunk rather than added piece by piece: the
53
+ // block carries no cache breakpoint, so there is nothing to gain from keeping the
54
+ // parts separable, and a single contiguous section reads as one instruction to the
55
+ // model instead of a pile of loose fragments.
56
+ const parts = [
57
+ ...sortSkills(ctx.resolve(ctx.input.callSkills ?? [])).map(renderSkill),
58
+ ...(ctx.input.context ?? []),
59
+ ]
60
+ if (parts.length > 0) {
61
+ ctx.add(PromptBlock.Context, joinChunks(parts))
62
+ }
63
+ },
64
+ }
65
+
66
+ /** The plugins every {@link PromptService} starts with, in run order. */
67
+ export const BUILT_IN_PROMPT_PLUGINS: readonly LlmPromptPlugin[] = [
68
+ rolePlugin, skillsPlugin, contextPlugin,
69
+ ] as const
@@ -0,0 +1,48 @@
1
+ import { DEFAULT_SKILL_ORDER } from '@owlmeans/llm-common'
2
+ import type { SkillDefinition } from '@owlmeans/llm-common'
3
+
4
+ /**
5
+ * Separator between rendered chunks. Every join in this file goes through it: the
6
+ * composed prompt must be byte-identical between calls, so there is exactly one way to
7
+ * glue things together.
8
+ */
9
+ export const CHUNK_SEPARATOR = '\n\n'
10
+
11
+ /**
12
+ * Code-unit comparison, NOT `localeCompare`.
13
+ *
14
+ * `localeCompare` orders differently depending on the host's ICU data and locale, so two
15
+ * processes could render the same skill set in different orders and never share a cache
16
+ * entry. Skill aliases are ASCII slugs; a plain comparison is both correct and stable.
17
+ */
18
+ export const compareAlias = (a: string, b: string): number => a < b ? -1 : a > b ? 1 : 0
19
+
20
+ /** Deterministic skill order: declared weight first, alias as the tiebreaker. */
21
+ export const sortSkills = (skills: readonly SkillDefinition[]): SkillDefinition[] =>
22
+ [...skills].sort((a, b) => {
23
+ const left = a.order ?? DEFAULT_SKILL_ORDER
24
+ const right = b.order ?? DEFAULT_SKILL_ORDER
25
+ return left !== right ? left - right : compareAlias(a.alias, b.alias)
26
+ })
27
+
28
+ /** Fixed rendering of one skill. Changing this shape invalidates every cached prefix. */
29
+ export const renderSkill = (skill: SkillDefinition): string =>
30
+ `## ${skill.title ?? skill.alias}\n\n${skill.body.trim()}`
31
+
32
+ /** Join rendered chunks into one block, dropping empties. */
33
+ export const joinChunks = (parts: readonly string[]): string =>
34
+ parts.map(part => part.trim()).filter(part => part !== '').join(CHUNK_SEPARATOR)
35
+
36
+ /**
37
+ * Stable digest of a cache prefix — FNV-1a, so there is no crypto dependency and the
38
+ * result is identical on every runtime. Used as a provider cache-routing key (OpenAI's
39
+ * `prompt_cache_key`), never for security.
40
+ */
41
+ export const prefixHash = (text: string): string => {
42
+ let hash = 0x811c9dc5
43
+ for (let i = 0; i < text.length; i++) {
44
+ hash ^= text.charCodeAt(i)
45
+ hash = Math.imul(hash, 0x01000193) >>> 0
46
+ }
47
+ return hash.toString(36)
48
+ }
@@ -0,0 +1,194 @@
1
+ import { createService } from '@owlmeans/context'
2
+ import type { BasicConfig, BasicContext } from '@owlmeans/context'
3
+ import { PROMPT_BLOCK_ORDER, PromptBlock } from '@owlmeans/llm-common'
4
+ import type { SkillDefinition } from '@owlmeans/llm-common'
5
+ import {
6
+ DEFAULT_CACHE_TTL, MAX_CACHE_BREAKPOINTS, MAX_SYSTEM_BREAKPOINTS, PROMPT_SERVICE,
7
+ } from '../consts.js'
8
+ import type { LlmSystemBlock } from '../plugins/types.js'
9
+ import { BUILT_IN_PROMPT_PLUGINS } from './plugins.js'
10
+ import { CHUNK_SEPARATOR, compareAlias, joinChunks } from './render.js'
11
+ import type {
12
+ LlmPromptPlugin, PromptContext, PromptResult, PromptService, PromptServiceOptions,
13
+ WithPromptService,
14
+ } from './types.js'
15
+
16
+ /** Sort weight of a plugin that declares none — between the built-in skills and context. */
17
+ const DEFAULT_PLUGIN_ORDER = 50
18
+
19
+ /** The part of {@link PromptService} this package implements — see {@link promptServiceApi}. */
20
+ export type PromptServiceApi =
21
+ Pick<PromptService, 'use' | 'register' | 'has' | 'resolve' | 'skills' | 'compose'>
22
+
23
+ /**
24
+ * Build the skill registry and composition chain WITHOUT registering a context service,
25
+ * so a consumer can spread it into its own `createService` and publish extra methods
26
+ * alongside it — the same pattern as `llmServiceApi` / `executionServiceApi`.
27
+ *
28
+ * `self` is late-bound because plugins resolve skills through the finished service, which
29
+ * a consumer may have extended.
30
+ */
31
+ export const promptServiceApi = (
32
+ options: PromptServiceOptions,
33
+ self: () => PromptService,
34
+ ): PromptServiceApi => {
35
+ const registry = new Map<string, SkillDefinition>()
36
+ /** Registration index per plugin alias — the stable tiebreaker for equal `order`. */
37
+ const seats = new Map<string, { plugin: LlmPromptPlugin; index: number }>()
38
+ let seq = 0
39
+
40
+ const seat = (plugin: LlmPromptPlugin): void => {
41
+ const existing = seats.get(plugin.alias)
42
+ // Re-registering under the same alias REPLACES rather than appends, so wiring the
43
+ // same plugin twice (a shared context builder plus an app) cannot double-emit.
44
+ seats.set(plugin.alias, { plugin, index: existing?.index ?? seq++ })
45
+ }
46
+
47
+ for (const plugin of [...BUILT_IN_PROMPT_PLUGINS, ...(options.plugins ?? [])]) {
48
+ seat(plugin)
49
+ }
50
+
51
+ const ordered = (): LlmPromptPlugin[] =>
52
+ [...seats.values()]
53
+ .sort((a, b) => {
54
+ const left = a.plugin.order ?? DEFAULT_PLUGIN_ORDER
55
+ const right = b.plugin.order ?? DEFAULT_PLUGIN_ORDER
56
+ return left !== right ? left - right : a.index - b.index
57
+ })
58
+ .map(entry => entry.plugin)
59
+
60
+ const api: PromptServiceApi = {
61
+
62
+ use: plugin => {
63
+ seat(plugin)
64
+ },
65
+
66
+ register: (...skills) => {
67
+ for (const skill of skills) {
68
+ registry.set(skill.alias, skill)
69
+ }
70
+ },
71
+
72
+ has: alias => registry.has(alias),
73
+
74
+ skills: () => [...registry.values()].sort((a, b) => compareAlias(a.alias, b.alias)),
75
+
76
+ /**
77
+ * Depth-first over `requires` so a dependency is emitted before the skill that pulled
78
+ * it in. Unknown aliases are skipped rather than thrown: a skill catalogue is often
79
+ * assembled from several packages and a missing optional one should degrade the
80
+ * prompt, not break the call.
81
+ */
82
+ resolve: aliases => {
83
+ const seen = new Set<string>()
84
+ const out: SkillDefinition[] = []
85
+ const walk = (alias: string): void => {
86
+ if (seen.has(alias)) {
87
+ return
88
+ }
89
+ seen.add(alias)
90
+ const skill = registry.get(alias)
91
+ if (skill == null) {
92
+ return
93
+ }
94
+ for (const required of skill.requires ?? []) {
95
+ walk(required)
96
+ }
97
+ out.push(skill)
98
+ }
99
+ for (const alias of aliases) {
100
+ walk(alias)
101
+ }
102
+
103
+ return out
104
+ },
105
+
106
+ compose: async (input, messages, params): Promise<PromptResult> => {
107
+ const sections = new Map<PromptBlock, string[]>()
108
+ const ctx: PromptContext = {
109
+ ...params,
110
+ input,
111
+ messages,
112
+ add: (block, text) => {
113
+ const trimmed = text.trim()
114
+ if (trimmed === '') {
115
+ return
116
+ }
117
+ const chunks = sections.get(block)
118
+ if (chunks == null) {
119
+ sections.set(block, [trimmed])
120
+ } else {
121
+ chunks.push(trimmed)
122
+ }
123
+ },
124
+ resolve: aliases => self().resolve(aliases),
125
+ }
126
+
127
+ // Two passes, not one: every static contribution must be in place before a plugin
128
+ // that reacts to the messages runs, so detection can see what is already covered.
129
+ const chain = ordered()
130
+ for (const plugin of chain) {
131
+ await plugin.compose?.(ctx)
132
+ }
133
+ for (const plugin of chain) {
134
+ await plugin.inspect?.(ctx)
135
+ }
136
+
137
+ const blocks: LlmSystemBlock[] = []
138
+ for (const block of PROMPT_BLOCK_ORDER) {
139
+ const text = joinChunks(sections.get(block) ?? [])
140
+ if (text !== '') {
141
+ blocks.push({ block, text })
142
+ }
143
+ }
144
+ if (blocks.length === 0) {
145
+ return { system: null, breakpoints: 0, blocks }
146
+ }
147
+
148
+ const cacheSystem = input.cacheSystem ?? options.cacheSystem ?? true
149
+ const ttl = input.cacheTtl ?? options.cacheTtl ?? DEFAULT_CACHE_TTL
150
+ const budget = Math.min(params.cacheMax ?? MAX_CACHE_BREAKPOINTS, MAX_SYSTEM_BREAKPOINTS)
151
+ const render = cacheSystem && budget > 0
152
+ ? params.provider?.patchSystem?.(blocks, { model: params.model, cacheMax: budget, ttl })
153
+ : null
154
+
155
+ return {
156
+ system: {
157
+ role: 'system',
158
+ content: render?.content ?? blocks.map(block => block.text).join(CHUNK_SEPARATOR),
159
+ },
160
+ breakpoints: render?.breakpoints ?? 0,
161
+ blocks,
162
+ }
163
+ },
164
+ }
165
+
166
+ api.register(...(options.skills ?? []))
167
+
168
+ return api
169
+ }
170
+
171
+ export const makePromptService = (
172
+ options: PromptServiceOptions = {},
173
+ alias: string = PROMPT_SERVICE,
174
+ ): PromptService => {
175
+ const service: PromptService = createService<PromptService>(
176
+ alias, promptServiceApi(options, () => service) as PromptService
177
+ )
178
+
179
+ return service
180
+ }
181
+
182
+ export const appendPromptService = <C extends BasicConfig, T extends BasicContext<C>>(
183
+ ctx: T,
184
+ options: PromptServiceOptions = {},
185
+ alias: string = PROMPT_SERVICE,
186
+ ): T & WithPromptService => {
187
+ const context = ctx as T & WithPromptService
188
+
189
+ context.registerService(makePromptService(options, alias))
190
+
191
+ context.prompts = () => context.service<PromptService>(alias)
192
+
193
+ return context
194
+ }
@@ -0,0 +1,114 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
+ import type { MessageFieldWithRole } from '@langchain/core/messages'
3
+ import type { InitializedService } from '@owlmeans/context'
4
+ import type {
5
+ CacheTtl, FileProviderRef, LlmPurpose, PromptBlock, PromptPolicy, SkillDefinition,
6
+ } from '@owlmeans/llm-common'
7
+ import type { LlmPlugin, LlmSystemBlock } from '../plugins/types.js'
8
+
9
+ /**
10
+ * Everything that shapes one composed system prompt. The serializable part
11
+ * ({@link PromptPolicy}) travels on the execution state; the rest is per-call.
12
+ */
13
+ export interface PromptInput extends PromptPolicy {
14
+ /** Skills supplied verbatim, without registering them first. */
15
+ inline?: SkillDefinition[]
16
+ /**
17
+ * Raw text for the volatile `Context` block — a caller's own system message, carried
18
+ * through unchanged. Never part of the cached prefix.
19
+ */
20
+ context?: string[]
21
+ /**
22
+ * Skill aliases requested for THIS call only. Rendered into `Context`, not `Skills`:
23
+ * a set that changes per call must not sit inside the region other calls cache.
24
+ */
25
+ callSkills?: string[]
26
+ }
27
+
28
+ export interface PromptComposeParams {
29
+ /** The ORIGINAL (unrefined) model the composed prompt will be sent to. */
30
+ model: BaseChatModel
31
+ /** Provider plugin governing the call — supplies the cache-marker strategy. */
32
+ provider?: LlmPlugin
33
+ purpose?: LlmPurpose
34
+ action?: string
35
+ /** Total breakpoint budget for the request. */
36
+ cacheMax?: number
37
+ /** File access a plugin may use to resolve knowledge from disk. */
38
+ files?: FileProviderRef
39
+ }
40
+
41
+ /** What a prompt plugin sees and may contribute to. */
42
+ export interface PromptContext extends PromptComposeParams {
43
+ input: PromptInput
44
+ /** The call's messages. Read them for detection; do NOT mutate. */
45
+ messages: readonly MessageFieldWithRole[]
46
+ /** Append a rendered chunk to a block. Empty text is ignored. */
47
+ add: (block: PromptBlock, text: string) => void
48
+ /** Resolve skill aliases through the registry, following `requires`. */
49
+ resolve: (aliases: readonly string[]) => SkillDefinition[]
50
+ }
51
+
52
+ /**
53
+ * A unit of system-prompt composition, registered on the {@link PromptService} at context
54
+ * composition time. This is the seam that makes the LLM layer pluggable: the package
55
+ * ships the role and skills plugins, an application adds its own.
56
+ *
57
+ * A plugin MUST be deterministic — same input, same bytes. Anything it contributes to a
58
+ * cached block and cannot reproduce exactly invalidates the prefix for every call that
59
+ * shares it.
60
+ */
61
+ export interface LlmPromptPlugin {
62
+ alias: string
63
+ /** Lower runs first. The built-ins occupy 0 (role), 10 (skills) and 90 (context). */
64
+ order?: number
65
+ /** Contribute static content, before anything has looked at the messages. */
66
+ compose?: (ctx: PromptContext) => void | Promise<void>
67
+ /** Contribute content derived from the messages (detection, lookup, fetch). */
68
+ inspect?: (ctx: PromptContext) => void | Promise<void>
69
+ }
70
+
71
+ export interface PromptResult {
72
+ /** The composed system message, or `null` when nothing was contributed. */
73
+ system: MessageFieldWithRole | null
74
+ /** Breakpoints the system prompt consumed; subtract from the message budget. */
75
+ breakpoints: number
76
+ /** The rendered blocks in emission order — exposed for tests and diagnostics. */
77
+ blocks: LlmSystemBlock[]
78
+ }
79
+
80
+ export interface PromptServiceOptions {
81
+ /** Skills registered up front; equivalent to calling `register` after construction. */
82
+ skills?: SkillDefinition[]
83
+ /** Plugins registered up front, in addition to the built-ins. */
84
+ plugins?: LlmPromptPlugin[]
85
+ /** Default for `PromptPolicy.cacheSystem`. Defaults to `true`. */
86
+ cacheSystem?: boolean
87
+ /** Default for `PromptPolicy.cacheTtl`. */
88
+ cacheTtl?: CacheTtl
89
+ }
90
+
91
+ /**
92
+ * Registry of reusable skills plus the plugin chain that turns a role and a set of skill
93
+ * aliases into a cache-friendly system prompt.
94
+ */
95
+ export interface PromptService extends InitializedService {
96
+ /** Register a composition plugin. Runs in `order` then registration order. */
97
+ use: (plugin: LlmPromptPlugin) => void
98
+ /** Register (or replace, by alias) reusable skills. */
99
+ register: (...skills: SkillDefinition[]) => void
100
+ has: (alias: string) => boolean
101
+ /** Resolve aliases to definitions, pulling in `requires` transitively. Unknown aliases are skipped. */
102
+ resolve: (aliases: readonly string[]) => SkillDefinition[]
103
+ /** Every registered skill, in deterministic order. */
104
+ skills: () => SkillDefinition[]
105
+ compose: (
106
+ input: PromptInput,
107
+ messages: readonly MessageFieldWithRole[],
108
+ params: PromptComposeParams,
109
+ ) => Promise<PromptResult>
110
+ }
111
+
112
+ export interface WithPromptService {
113
+ prompts: () => PromptService
114
+ }
package/src/service.ts CHANGED
@@ -51,6 +51,9 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
51
51
  throw new LlmMissconfiguredError(alias)
52
52
  }
53
53
  const config: ModelConfig = { ...baseConfig, ...override }
54
+ // The service-wide idle deadline is a floor, not an override: a preset that states its
55
+ // own `streamTimeout` knows something specific about that model and keeps it.
56
+ config.streamTimeout ??= options.streamTimeout
54
57
  const preset: Partial<ModelConfig> = config.preset != null
55
58
  ? { ...(models.find(m => m.alias === config.preset) ?? {}) }
56
59
  : {}
package/src/types.ts CHANGED
@@ -4,8 +4,10 @@ import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/cor
4
4
  import type { JSONSchemaType } from 'ajv'
5
5
  import type { InitializedService } from '@owlmeans/context'
6
6
  import type {
7
- LlmPurpose, ModelProvider, NullCapture, SpectatorArgument, SpectatorEntryLogged,
7
+ FileProviderRef, LlmPurpose, ModelProvider, NullCapture, SpectatorArgument,
8
+ SpectatorEntryLogged,
8
9
  } from '@owlmeans/llm-common'
10
+ import type { PromptInput, PromptService } from './prompt/types.js'
9
11
 
10
12
  export type MaybeArray<T> = T | T[]
11
13
 
@@ -47,6 +49,20 @@ export interface LlmLogging {
47
49
  export interface LlmModelOptions extends LlmLogging {
48
50
  model: BaseChatModel
49
51
  retries?: number
52
+ /**
53
+ * Role and skills for every call this model makes. Composed into a cacheable system
54
+ * prompt by {@link PromptService}; ignored when no `prompts` resolver is supplied.
55
+ * Usually just `exec.prompt` from the helper execution.
56
+ */
57
+ prompt?: PromptInput
58
+ /**
59
+ * Late-bound resolver for the prompt service — a function, like `Execution.models`, so
60
+ * the service can be swapped or cloned. Without it the model behaves exactly as before
61
+ * this layer existed: the caller's messages are sent untouched.
62
+ */
63
+ prompts?: () => PromptService
64
+ /** File access offered to prompt plugins that resolve knowledge from disk. */
65
+ files?: FileProviderRef
50
66
  }
51
67
 
52
68
  export type ModelMessage = BaseMessage | MessageFieldWithRole
@@ -56,10 +72,23 @@ export type ModelInput = MaybeArray<ModelInputItem>
56
72
  export interface LlmCallOptions {
57
73
  /** Short name of the operation — used as the LangChain run name and in spectator entries. */
58
74
  action: string
59
- /** Ask the provider to cache the prompt prefix (no-op for providers without prompt caching). */
75
+ /**
76
+ * Cache the leading MESSAGES too. The composed system prompt is cached by default and
77
+ * independently of this flag — this one is about the conversation prefix, which is only
78
+ * worth caching when the same leading messages recur across calls.
79
+ */
60
80
  useCache?: boolean
61
- /** How many leading messages to mark as cacheable (capped by the provider's own limit). */
81
+ /**
82
+ * How many leading messages form the stable prefix. One breakpoint is placed at its
83
+ * end (not one per message), capped by whatever the system prompt left unspent.
84
+ */
62
85
  cacheMax?: number
86
+ /**
87
+ * Skill aliases for THIS call only. Rendered into the volatile `Context` block, so they
88
+ * never disturb the cached region — declare a skill on the execution instead when it
89
+ * should be part of the shared prefix.
90
+ */
91
+ skills?: string[]
63
92
  }
64
93
 
65
94
  export interface LlmAskOptions extends LlmCallOptions {
@@ -183,11 +212,33 @@ export interface ModelConfig {
183
212
  * rotating providers mid-call would flip the structured-output format.
184
213
  */
185
214
  fallback?: Partial<ModelConfig>
215
+ /**
216
+ * Per-model override of {@link MIN_CACHEABLE_TOKENS} — the shortest prefix worth a
217
+ * cache breakpoint. Anthropic's own minimum is model-dependent and NOT monotonic
218
+ * across generations (512 on the newest, 1024 on most, 4096 on a few older ones), so a
219
+ * preset that pins an old model should raise this rather than pay for markers that
220
+ * silently never cache.
221
+ */
222
+ cacheMinTokens?: number
223
+ /**
224
+ * Cache-routing key for providers whose prompt cache is automatic (OpenAI's
225
+ * `prompt_cache_key`): requests sharing a key are routed to the same backend, which
226
+ * raises the hit rate for a shared prefix. Must be STABLE and low-cardinality — one
227
+ * value per role, never per user or per request. Defaults to the config alias.
228
+ */
229
+ cacheKey?: string
186
230
  }
187
231
 
188
232
  export interface LlmServiceOptions {
189
233
  /** The full config list; resolved by `alias` on every `getModel` call. */
190
234
  models: () => ModelConfig[]
235
+ /**
236
+ * Idle deadline (ms) applied to every model this service builds, unless the model's own
237
+ * config overrides it. This is the knob an application sets where it composes its
238
+ * context — one place to tune how long the whole deployment waits on a silent provider,
239
+ * without touching a preset. Falls back to {@link MODEL_STREAM_TIMEOUT_MS}.
240
+ */
241
+ streamTimeout?: number
191
242
  }
192
243
 
193
244
  /**