@owlmeans/llm 0.1.15 → 0.1.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -4
- package/agent-meta/manifest.json +9 -9
- package/agent-meta/skills/llm/SKILL.md +54 -4
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
- package/build/consts.d.ts +36 -3
- package/build/consts.d.ts.map +1 -1
- package/build/consts.js +37 -4
- package/build/consts.js.map +1 -1
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +8 -3
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +20 -1
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts +12 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +22 -0
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/cache.d.ts +17 -0
- package/build/helpers/cache.d.ts.map +1 -0
- package/build/helpers/cache.js +22 -0
- package/build/helpers/cache.js.map +1 -0
- package/build/helpers/index.d.ts +1 -0
- package/build/helpers/index.d.ts.map +1 -1
- package/build/helpers/index.js +1 -0
- package/build/helpers/index.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +7 -0
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +1 -0
- package/build/index.d.ts.map +1 -1
- package/build/index.js +1 -0
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +90 -16
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +136 -21
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +9 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +23 -1
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +38 -5
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/index.d.ts +5 -0
- package/build/prompt/index.d.ts.map +1 -0
- package/build/prompt/index.js +4 -0
- package/build/prompt/index.js.map +1 -0
- package/build/prompt/plugins.d.ts +24 -0
- package/build/prompt/plugins.d.ts.map +1 -0
- package/build/prompt/plugins.js +64 -0
- package/build/prompt/plugins.js.map +1 -0
- package/build/prompt/render.d.ts +28 -0
- package/build/prompt/render.d.ts.map +1 -0
- package/build/prompt/render.js +39 -0
- package/build/prompt/render.js.map +1 -0
- package/build/prompt/service.d.ts +16 -0
- package/build/prompt/service.d.ts.map +1 -0
- package/build/prompt/service.js +145 -0
- package/build/prompt/service.js.map +1 -0
- package/build/prompt/types.d.ts +101 -0
- package/build/prompt/types.d.ts.map +1 -0
- package/build/prompt/types.js +2 -0
- package/build/prompt/types.js.map +1 -0
- package/build/service.d.ts.map +1 -1
- package/build/service.js +3 -0
- package/build/service.js.map +1 -1
- package/build/types.d.ts +53 -3
- package/build/types.d.ts.map +1 -1
- package/build/utils/prompt.d.ts +14 -0
- package/build/utils/prompt.d.ts.map +1 -1
- package/build/utils/prompt.js +32 -0
- package/build/utils/prompt.js.map +1 -1
- package/package.json +13 -6
- package/src/consts.ts +43 -5
- package/src/execution/service.ts +9 -3
- package/src/execution/types.ts +21 -2
- package/src/execution/utils.ts +28 -1
- package/src/helpers/cache.ts +33 -0
- package/src/helpers/index.ts +1 -0
- package/src/helpers/spectate.ts +10 -0
- package/src/index.ts +1 -0
- package/src/model.ts +119 -16
- package/src/plugins/anthropic.ts +159 -20
- package/src/plugins/openai.ts +26 -1
- package/src/plugins/types.ts +42 -5
- package/src/prompt/index.ts +5 -0
- package/src/prompt/plugins.ts +69 -0
- package/src/prompt/render.ts +48 -0
- package/src/prompt/service.ts +194 -0
- package/src/prompt/types.ts +114 -0
- package/src/service.ts +3 -0
- package/src/types.ts +54 -3
- package/src/utils/prompt.ts +33 -0
- package/tests/execution.spec.ts +42 -0
- package/tests/plugins.spec.ts +279 -14
- package/tests/prompt.spec.ts +194 -0
- package/agent-meta/instructions/llm.instructions.md +0 -66
package/src/plugins/openai.ts
CHANGED
|
@@ -33,6 +33,17 @@ export const openAiFamily = {
|
|
|
33
33
|
json_schema: { name: toolName, schema, strict: false },
|
|
34
34
|
}),
|
|
35
35
|
|
|
36
|
+
/**
|
|
37
|
+
* A 400 means the request itself is malformed — a schema the endpoint rejects, an
|
|
38
|
+
* unsupported parameter, a `max_tokens` above the model's per-request limit. Retrying
|
|
39
|
+
* re-sends the same shape (and the escalator raises `max_tokens`, making the last case
|
|
40
|
+
* strictly worse), so eight attempts only bury the real message. Matched on `status`
|
|
41
|
+
* rather than an SDK class: aggregators and nested SDK copies throw their own error
|
|
42
|
+
* types, and an `instanceof` against one of them silently never matches.
|
|
43
|
+
*/
|
|
44
|
+
isFatal: (e: unknown): Error | null =>
|
|
45
|
+
(e as { status?: unknown })?.status === 400 ? e as Error : null,
|
|
46
|
+
|
|
36
47
|
refine: ({ base, attempt, temperature, maxOutputCap }: LlmRefineParams): BaseChatModel => {
|
|
37
48
|
const model = base as ChatOpenAI
|
|
38
49
|
const currentTemperature = temperature ?? model.temperature ?? 0
|
|
@@ -75,10 +86,22 @@ export const openAiPlugin: LlmPlugin = {
|
|
|
75
86
|
structuredMode: (config: ModelConfig): StructuredMode =>
|
|
76
87
|
config.structuredOutput === false ? StructuredMode.Tool : StructuredMode.Native,
|
|
77
88
|
|
|
78
|
-
build: ({ config, secret, callbacks }) => {
|
|
89
|
+
build: ({ alias, config, secret, callbacks }) => {
|
|
79
90
|
const model = config.model ??= 'gpt-5.4-mini'
|
|
80
91
|
const configuration = makeConfiguration({ baseURL: undefined, headers: config.headers })
|
|
81
92
|
|
|
93
|
+
// OpenAI's prompt cache is automatic and prefix-based — there is nothing to mark. The
|
|
94
|
+
// one lever a client has is routing: requests are dispatched by a hash of the prompt's
|
|
95
|
+
// opening tokens, and `prompt_cache_key` is mixed into that hash, so requests sharing
|
|
96
|
+
// a key land on the same backend and can actually hit each other's entries. The key
|
|
97
|
+
// must be stable and low-cardinality; the config alias IS the role, which is exactly
|
|
98
|
+
// the granularity at which a system prefix is shared.
|
|
99
|
+
//
|
|
100
|
+
// Deliberately NOT done for the `compatible` plugin: aggregators there run with
|
|
101
|
+
// `provider.require_parameters`, and an unknown top-level field can exclude every
|
|
102
|
+
// serving provider from the route.
|
|
103
|
+
const modelKwargs = { prompt_cache_key: config.cacheKey ?? alias }
|
|
104
|
+
|
|
82
105
|
// The Responses API models reject `temperature`/`topP`.
|
|
83
106
|
if (RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))) {
|
|
84
107
|
return new ChatOpenAI({
|
|
@@ -89,6 +112,7 @@ export const openAiPlugin: LlmPlugin = {
|
|
|
89
112
|
useResponsesApi: true,
|
|
90
113
|
metadata: { config },
|
|
91
114
|
callbacks,
|
|
115
|
+
modelKwargs,
|
|
92
116
|
...configuration,
|
|
93
117
|
})
|
|
94
118
|
}
|
|
@@ -102,6 +126,7 @@ export const openAiPlugin: LlmPlugin = {
|
|
|
102
126
|
maxRetries: 5,
|
|
103
127
|
metadata: { config },
|
|
104
128
|
callbacks,
|
|
129
|
+
modelKwargs,
|
|
105
130
|
...configuration,
|
|
106
131
|
})
|
|
107
132
|
},
|
package/src/plugins/types.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
2
|
import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/core/callbacks/base'
|
|
3
|
-
import type { MessageFieldWithRole } from '@langchain/core/messages'
|
|
4
|
-
import type { StructuredMode } from '@owlmeans/llm-common'
|
|
3
|
+
import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
|
|
4
|
+
import type { CacheTtl, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
|
|
5
5
|
import type { ModelConfig } from '../types.js'
|
|
6
6
|
|
|
7
7
|
export interface LlmBuildParams {
|
|
@@ -32,7 +32,32 @@ export interface LlmCacheParams {
|
|
|
32
32
|
/** The ORIGINAL (unrefined) model — refined instances do not always keep the name. */
|
|
33
33
|
model: BaseChatModel
|
|
34
34
|
useCache: boolean
|
|
35
|
+
/** How many leading messages form the stable prefix worth caching. */
|
|
35
36
|
cacheMax: number
|
|
37
|
+
/** Breakpoints the composed system prompt already spent on this request. */
|
|
38
|
+
reserved?: number
|
|
39
|
+
ttl?: CacheTtl
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** One rendered section of the composed system prompt, in emission order. */
|
|
43
|
+
export interface LlmSystemBlock {
|
|
44
|
+
block: PromptBlock
|
|
45
|
+
text: string
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface LlmSystemCacheParams {
|
|
49
|
+
/** The ORIGINAL (unrefined) model. */
|
|
50
|
+
model: BaseChatModel
|
|
51
|
+
/** Breakpoint budget the system prompt may spend. */
|
|
52
|
+
cacheMax: number
|
|
53
|
+
ttl: CacheTtl
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Provider-native rendering of the composed system prompt. */
|
|
57
|
+
export interface LlmSystemRender {
|
|
58
|
+
content: MessageContent
|
|
59
|
+
/** Breakpoints consumed — subtracted from the message-level budget. */
|
|
60
|
+
breakpoints: number
|
|
36
61
|
}
|
|
37
62
|
|
|
38
63
|
/**
|
|
@@ -78,12 +103,24 @@ export interface LlmPlugin {
|
|
|
78
103
|
responseFormat?: (toolName: string, schema: unknown) => Record<string, unknown>
|
|
79
104
|
|
|
80
105
|
/**
|
|
81
|
-
* Mark
|
|
82
|
-
*
|
|
83
|
-
*
|
|
106
|
+
* Mark the stable message prefix as cacheable, in-place. ONE breakpoint at the end of
|
|
107
|
+
* that prefix — not one per message: the budget is four per request and the system
|
|
108
|
+
* prompt has first claim on it. Returns `true` when a marker was actually placed.
|
|
109
|
+
* Omit for providers with no explicit prompt-cache markers.
|
|
84
110
|
*/
|
|
85
111
|
patchCache?: (msgs: MessageFieldWithRole[], params: LlmCacheParams) => boolean
|
|
86
112
|
|
|
113
|
+
/**
|
|
114
|
+
* Render the composed system blocks into provider-native content, placing cache
|
|
115
|
+
* breakpoints at the stability boundaries between blocks (see `PromptBlock`).
|
|
116
|
+
*
|
|
117
|
+
* Return `null` — or omit the method entirely — for providers whose prompt cache is
|
|
118
|
+
* automatic and prefix-based (OpenAI and friends): the service then joins the blocks
|
|
119
|
+
* into a plain string, which is all those providers need, since the block ORDER is
|
|
120
|
+
* what makes their prefix stable.
|
|
121
|
+
*/
|
|
122
|
+
patchSystem?: (blocks: LlmSystemBlock[], params: LlmSystemCacheParams) => LlmSystemRender | null
|
|
123
|
+
|
|
87
124
|
/**
|
|
88
125
|
* Classify a thrown error as fatal for the retry loop. Return the error to throw
|
|
89
126
|
* immediately, or `null` to let it be retried.
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import { PromptBlock } from '@owlmeans/llm-common'
|
|
2
|
+
import type { SkillDefinition } from '@owlmeans/llm-common'
|
|
3
|
+
import { joinChunks, renderSkill, sortSkills } from './render.js'
|
|
4
|
+
import type { LlmPromptPlugin } from './types.js'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Block 0 — the base system prompt that tells the model who it is.
|
|
8
|
+
*
|
|
9
|
+
* The most stable thing in the whole request, so it goes first and every cache boundary
|
|
10
|
+
* sits behind it.
|
|
11
|
+
*/
|
|
12
|
+
export const rolePlugin: LlmPromptPlugin = {
|
|
13
|
+
alias: 'role',
|
|
14
|
+
order: 0,
|
|
15
|
+
compose: ctx => {
|
|
16
|
+
if (ctx.input.role != null && ctx.input.role.trim() !== '') {
|
|
17
|
+
ctx.add(PromptBlock.Role, ctx.input.role)
|
|
18
|
+
}
|
|
19
|
+
},
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Block 1 — the declared capabilities, rendered in a deterministic order.
|
|
24
|
+
*
|
|
25
|
+
* Registry skills and inline skills are merged by alias with inline winning, so a caller
|
|
26
|
+
* can override one registered entry without forking the catalogue.
|
|
27
|
+
*/
|
|
28
|
+
export const skillsPlugin: LlmPromptPlugin = {
|
|
29
|
+
alias: 'skills',
|
|
30
|
+
order: 10,
|
|
31
|
+
compose: ctx => {
|
|
32
|
+
const declared = ctx.resolve(ctx.input.skills ?? [])
|
|
33
|
+
const merged = new Map<string, SkillDefinition>()
|
|
34
|
+
for (const skill of [...declared, ...(ctx.input.inline ?? [])]) {
|
|
35
|
+
merged.set(skill.alias, skill)
|
|
36
|
+
}
|
|
37
|
+
for (const skill of sortSkills([...merged.values()])) {
|
|
38
|
+
ctx.add(skill.block ?? PromptBlock.Skills, renderSkill(skill))
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Block 3 — whatever the caller handed over verbatim, plus any skills requested for this
|
|
45
|
+
* one call. Emitted last and never marked cacheable: its content varies per request by
|
|
46
|
+
* definition, and a varying tail must not sit inside a prefix other calls depend on.
|
|
47
|
+
*/
|
|
48
|
+
export const contextPlugin: LlmPromptPlugin = {
|
|
49
|
+
alias: 'context',
|
|
50
|
+
order: 90,
|
|
51
|
+
compose: ctx => {
|
|
52
|
+
// Everything volatile is merged into ONE chunk rather than added piece by piece: the
|
|
53
|
+
// block carries no cache breakpoint, so there is nothing to gain from keeping the
|
|
54
|
+
// parts separable, and a single contiguous section reads as one instruction to the
|
|
55
|
+
// model instead of a pile of loose fragments.
|
|
56
|
+
const parts = [
|
|
57
|
+
...sortSkills(ctx.resolve(ctx.input.callSkills ?? [])).map(renderSkill),
|
|
58
|
+
...(ctx.input.context ?? []),
|
|
59
|
+
]
|
|
60
|
+
if (parts.length > 0) {
|
|
61
|
+
ctx.add(PromptBlock.Context, joinChunks(parts))
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** The plugins every {@link PromptService} starts with, in run order. */
|
|
67
|
+
export const BUILT_IN_PROMPT_PLUGINS: readonly LlmPromptPlugin[] = [
|
|
68
|
+
rolePlugin, skillsPlugin, contextPlugin,
|
|
69
|
+
] as const
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import { DEFAULT_SKILL_ORDER } from '@owlmeans/llm-common'
|
|
2
|
+
import type { SkillDefinition } from '@owlmeans/llm-common'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Separator between rendered chunks. Every join in this file goes through it: the
|
|
6
|
+
* composed prompt must be byte-identical between calls, so there is exactly one way to
|
|
7
|
+
* glue things together.
|
|
8
|
+
*/
|
|
9
|
+
export const CHUNK_SEPARATOR = '\n\n'
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Code-unit comparison, NOT `localeCompare`.
|
|
13
|
+
*
|
|
14
|
+
* `localeCompare` orders differently depending on the host's ICU data and locale, so two
|
|
15
|
+
* processes could render the same skill set in different orders and never share a cache
|
|
16
|
+
* entry. Skill aliases are ASCII slugs; a plain comparison is both correct and stable.
|
|
17
|
+
*/
|
|
18
|
+
export const compareAlias = (a: string, b: string): number => a < b ? -1 : a > b ? 1 : 0
|
|
19
|
+
|
|
20
|
+
/** Deterministic skill order: declared weight first, alias as the tiebreaker. */
|
|
21
|
+
export const sortSkills = (skills: readonly SkillDefinition[]): SkillDefinition[] =>
|
|
22
|
+
[...skills].sort((a, b) => {
|
|
23
|
+
const left = a.order ?? DEFAULT_SKILL_ORDER
|
|
24
|
+
const right = b.order ?? DEFAULT_SKILL_ORDER
|
|
25
|
+
return left !== right ? left - right : compareAlias(a.alias, b.alias)
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
/** Fixed rendering of one skill. Changing this shape invalidates every cached prefix. */
|
|
29
|
+
export const renderSkill = (skill: SkillDefinition): string =>
|
|
30
|
+
`## ${skill.title ?? skill.alias}\n\n${skill.body.trim()}`
|
|
31
|
+
|
|
32
|
+
/** Join rendered chunks into one block, dropping empties. */
|
|
33
|
+
export const joinChunks = (parts: readonly string[]): string =>
|
|
34
|
+
parts.map(part => part.trim()).filter(part => part !== '').join(CHUNK_SEPARATOR)
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Stable digest of a cache prefix — FNV-1a, so there is no crypto dependency and the
|
|
38
|
+
* result is identical on every runtime. Used as a provider cache-routing key (OpenAI's
|
|
39
|
+
* `prompt_cache_key`), never for security.
|
|
40
|
+
*/
|
|
41
|
+
export const prefixHash = (text: string): string => {
|
|
42
|
+
let hash = 0x811c9dc5
|
|
43
|
+
for (let i = 0; i < text.length; i++) {
|
|
44
|
+
hash ^= text.charCodeAt(i)
|
|
45
|
+
hash = Math.imul(hash, 0x01000193) >>> 0
|
|
46
|
+
}
|
|
47
|
+
return hash.toString(36)
|
|
48
|
+
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
import { createService } from '@owlmeans/context'
|
|
2
|
+
import type { BasicConfig, BasicContext } from '@owlmeans/context'
|
|
3
|
+
import { PROMPT_BLOCK_ORDER, PromptBlock } from '@owlmeans/llm-common'
|
|
4
|
+
import type { SkillDefinition } from '@owlmeans/llm-common'
|
|
5
|
+
import {
|
|
6
|
+
DEFAULT_CACHE_TTL, MAX_CACHE_BREAKPOINTS, MAX_SYSTEM_BREAKPOINTS, PROMPT_SERVICE,
|
|
7
|
+
} from '../consts.js'
|
|
8
|
+
import type { LlmSystemBlock } from '../plugins/types.js'
|
|
9
|
+
import { BUILT_IN_PROMPT_PLUGINS } from './plugins.js'
|
|
10
|
+
import { CHUNK_SEPARATOR, compareAlias, joinChunks } from './render.js'
|
|
11
|
+
import type {
|
|
12
|
+
LlmPromptPlugin, PromptContext, PromptResult, PromptService, PromptServiceOptions,
|
|
13
|
+
WithPromptService,
|
|
14
|
+
} from './types.js'
|
|
15
|
+
|
|
16
|
+
/** Sort weight of a plugin that declares none — between the built-in skills and context. */
|
|
17
|
+
const DEFAULT_PLUGIN_ORDER = 50
|
|
18
|
+
|
|
19
|
+
/** The part of {@link PromptService} this package implements — see {@link promptServiceApi}. */
|
|
20
|
+
export type PromptServiceApi =
|
|
21
|
+
Pick<PromptService, 'use' | 'register' | 'has' | 'resolve' | 'skills' | 'compose'>
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Build the skill registry and composition chain WITHOUT registering a context service,
|
|
25
|
+
* so a consumer can spread it into its own `createService` and publish extra methods
|
|
26
|
+
* alongside it — the same pattern as `llmServiceApi` / `executionServiceApi`.
|
|
27
|
+
*
|
|
28
|
+
* `self` is late-bound because plugins resolve skills through the finished service, which
|
|
29
|
+
* a consumer may have extended.
|
|
30
|
+
*/
|
|
31
|
+
export const promptServiceApi = (
|
|
32
|
+
options: PromptServiceOptions,
|
|
33
|
+
self: () => PromptService,
|
|
34
|
+
): PromptServiceApi => {
|
|
35
|
+
const registry = new Map<string, SkillDefinition>()
|
|
36
|
+
/** Registration index per plugin alias — the stable tiebreaker for equal `order`. */
|
|
37
|
+
const seats = new Map<string, { plugin: LlmPromptPlugin; index: number }>()
|
|
38
|
+
let seq = 0
|
|
39
|
+
|
|
40
|
+
const seat = (plugin: LlmPromptPlugin): void => {
|
|
41
|
+
const existing = seats.get(plugin.alias)
|
|
42
|
+
// Re-registering under the same alias REPLACES rather than appends, so wiring the
|
|
43
|
+
// same plugin twice (a shared context builder plus an app) cannot double-emit.
|
|
44
|
+
seats.set(plugin.alias, { plugin, index: existing?.index ?? seq++ })
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
for (const plugin of [...BUILT_IN_PROMPT_PLUGINS, ...(options.plugins ?? [])]) {
|
|
48
|
+
seat(plugin)
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const ordered = (): LlmPromptPlugin[] =>
|
|
52
|
+
[...seats.values()]
|
|
53
|
+
.sort((a, b) => {
|
|
54
|
+
const left = a.plugin.order ?? DEFAULT_PLUGIN_ORDER
|
|
55
|
+
const right = b.plugin.order ?? DEFAULT_PLUGIN_ORDER
|
|
56
|
+
return left !== right ? left - right : a.index - b.index
|
|
57
|
+
})
|
|
58
|
+
.map(entry => entry.plugin)
|
|
59
|
+
|
|
60
|
+
const api: PromptServiceApi = {
|
|
61
|
+
|
|
62
|
+
use: plugin => {
|
|
63
|
+
seat(plugin)
|
|
64
|
+
},
|
|
65
|
+
|
|
66
|
+
register: (...skills) => {
|
|
67
|
+
for (const skill of skills) {
|
|
68
|
+
registry.set(skill.alias, skill)
|
|
69
|
+
}
|
|
70
|
+
},
|
|
71
|
+
|
|
72
|
+
has: alias => registry.has(alias),
|
|
73
|
+
|
|
74
|
+
skills: () => [...registry.values()].sort((a, b) => compareAlias(a.alias, b.alias)),
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Depth-first over `requires` so a dependency is emitted before the skill that pulled
|
|
78
|
+
* it in. Unknown aliases are skipped rather than thrown: a skill catalogue is often
|
|
79
|
+
* assembled from several packages and a missing optional one should degrade the
|
|
80
|
+
* prompt, not break the call.
|
|
81
|
+
*/
|
|
82
|
+
resolve: aliases => {
|
|
83
|
+
const seen = new Set<string>()
|
|
84
|
+
const out: SkillDefinition[] = []
|
|
85
|
+
const walk = (alias: string): void => {
|
|
86
|
+
if (seen.has(alias)) {
|
|
87
|
+
return
|
|
88
|
+
}
|
|
89
|
+
seen.add(alias)
|
|
90
|
+
const skill = registry.get(alias)
|
|
91
|
+
if (skill == null) {
|
|
92
|
+
return
|
|
93
|
+
}
|
|
94
|
+
for (const required of skill.requires ?? []) {
|
|
95
|
+
walk(required)
|
|
96
|
+
}
|
|
97
|
+
out.push(skill)
|
|
98
|
+
}
|
|
99
|
+
for (const alias of aliases) {
|
|
100
|
+
walk(alias)
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
return out
|
|
104
|
+
},
|
|
105
|
+
|
|
106
|
+
compose: async (input, messages, params): Promise<PromptResult> => {
|
|
107
|
+
const sections = new Map<PromptBlock, string[]>()
|
|
108
|
+
const ctx: PromptContext = {
|
|
109
|
+
...params,
|
|
110
|
+
input,
|
|
111
|
+
messages,
|
|
112
|
+
add: (block, text) => {
|
|
113
|
+
const trimmed = text.trim()
|
|
114
|
+
if (trimmed === '') {
|
|
115
|
+
return
|
|
116
|
+
}
|
|
117
|
+
const chunks = sections.get(block)
|
|
118
|
+
if (chunks == null) {
|
|
119
|
+
sections.set(block, [trimmed])
|
|
120
|
+
} else {
|
|
121
|
+
chunks.push(trimmed)
|
|
122
|
+
}
|
|
123
|
+
},
|
|
124
|
+
resolve: aliases => self().resolve(aliases),
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// Two passes, not one: every static contribution must be in place before a plugin
|
|
128
|
+
// that reacts to the messages runs, so detection can see what is already covered.
|
|
129
|
+
const chain = ordered()
|
|
130
|
+
for (const plugin of chain) {
|
|
131
|
+
await plugin.compose?.(ctx)
|
|
132
|
+
}
|
|
133
|
+
for (const plugin of chain) {
|
|
134
|
+
await plugin.inspect?.(ctx)
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
const blocks: LlmSystemBlock[] = []
|
|
138
|
+
for (const block of PROMPT_BLOCK_ORDER) {
|
|
139
|
+
const text = joinChunks(sections.get(block) ?? [])
|
|
140
|
+
if (text !== '') {
|
|
141
|
+
blocks.push({ block, text })
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
if (blocks.length === 0) {
|
|
145
|
+
return { system: null, breakpoints: 0, blocks }
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const cacheSystem = input.cacheSystem ?? options.cacheSystem ?? true
|
|
149
|
+
const ttl = input.cacheTtl ?? options.cacheTtl ?? DEFAULT_CACHE_TTL
|
|
150
|
+
const budget = Math.min(params.cacheMax ?? MAX_CACHE_BREAKPOINTS, MAX_SYSTEM_BREAKPOINTS)
|
|
151
|
+
const render = cacheSystem && budget > 0
|
|
152
|
+
? params.provider?.patchSystem?.(blocks, { model: params.model, cacheMax: budget, ttl })
|
|
153
|
+
: null
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
system: {
|
|
157
|
+
role: 'system',
|
|
158
|
+
content: render?.content ?? blocks.map(block => block.text).join(CHUNK_SEPARATOR),
|
|
159
|
+
},
|
|
160
|
+
breakpoints: render?.breakpoints ?? 0,
|
|
161
|
+
blocks,
|
|
162
|
+
}
|
|
163
|
+
},
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
api.register(...(options.skills ?? []))
|
|
167
|
+
|
|
168
|
+
return api
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
export const makePromptService = (
|
|
172
|
+
options: PromptServiceOptions = {},
|
|
173
|
+
alias: string = PROMPT_SERVICE,
|
|
174
|
+
): PromptService => {
|
|
175
|
+
const service: PromptService = createService<PromptService>(
|
|
176
|
+
alias, promptServiceApi(options, () => service) as PromptService
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
return service
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
export const appendPromptService = <C extends BasicConfig, T extends BasicContext<C>>(
|
|
183
|
+
ctx: T,
|
|
184
|
+
options: PromptServiceOptions = {},
|
|
185
|
+
alias: string = PROMPT_SERVICE,
|
|
186
|
+
): T & WithPromptService => {
|
|
187
|
+
const context = ctx as T & WithPromptService
|
|
188
|
+
|
|
189
|
+
context.registerService(makePromptService(options, alias))
|
|
190
|
+
|
|
191
|
+
context.prompts = () => context.service<PromptService>(alias)
|
|
192
|
+
|
|
193
|
+
return context
|
|
194
|
+
}
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
|
+
import type { MessageFieldWithRole } from '@langchain/core/messages'
|
|
3
|
+
import type { InitializedService } from '@owlmeans/context'
|
|
4
|
+
import type {
|
|
5
|
+
CacheTtl, FileProviderRef, LlmPurpose, PromptBlock, PromptPolicy, SkillDefinition,
|
|
6
|
+
} from '@owlmeans/llm-common'
|
|
7
|
+
import type { LlmPlugin, LlmSystemBlock } from '../plugins/types.js'
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Everything that shapes one composed system prompt. The serializable part
|
|
11
|
+
* ({@link PromptPolicy}) travels on the execution state; the rest is per-call.
|
|
12
|
+
*/
|
|
13
|
+
export interface PromptInput extends PromptPolicy {
|
|
14
|
+
/** Skills supplied verbatim, without registering them first. */
|
|
15
|
+
inline?: SkillDefinition[]
|
|
16
|
+
/**
|
|
17
|
+
* Raw text for the volatile `Context` block — a caller's own system message, carried
|
|
18
|
+
* through unchanged. Never part of the cached prefix.
|
|
19
|
+
*/
|
|
20
|
+
context?: string[]
|
|
21
|
+
/**
|
|
22
|
+
* Skill aliases requested for THIS call only. Rendered into `Context`, not `Skills`:
|
|
23
|
+
* a set that changes per call must not sit inside the region other calls cache.
|
|
24
|
+
*/
|
|
25
|
+
callSkills?: string[]
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export interface PromptComposeParams {
|
|
29
|
+
/** The ORIGINAL (unrefined) model the composed prompt will be sent to. */
|
|
30
|
+
model: BaseChatModel
|
|
31
|
+
/** Provider plugin governing the call — supplies the cache-marker strategy. */
|
|
32
|
+
provider?: LlmPlugin
|
|
33
|
+
purpose?: LlmPurpose
|
|
34
|
+
action?: string
|
|
35
|
+
/** Total breakpoint budget for the request. */
|
|
36
|
+
cacheMax?: number
|
|
37
|
+
/** File access a plugin may use to resolve knowledge from disk. */
|
|
38
|
+
files?: FileProviderRef
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** What a prompt plugin sees and may contribute to. */
|
|
42
|
+
export interface PromptContext extends PromptComposeParams {
|
|
43
|
+
input: PromptInput
|
|
44
|
+
/** The call's messages. Read them for detection; do NOT mutate. */
|
|
45
|
+
messages: readonly MessageFieldWithRole[]
|
|
46
|
+
/** Append a rendered chunk to a block. Empty text is ignored. */
|
|
47
|
+
add: (block: PromptBlock, text: string) => void
|
|
48
|
+
/** Resolve skill aliases through the registry, following `requires`. */
|
|
49
|
+
resolve: (aliases: readonly string[]) => SkillDefinition[]
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* A unit of system-prompt composition, registered on the {@link PromptService} at context
|
|
54
|
+
* composition time. This is the seam that makes the LLM layer pluggable: the package
|
|
55
|
+
* ships the role and skills plugins, an application adds its own.
|
|
56
|
+
*
|
|
57
|
+
* A plugin MUST be deterministic — same input, same bytes. Anything it contributes to a
|
|
58
|
+
* cached block and cannot reproduce exactly invalidates the prefix for every call that
|
|
59
|
+
* shares it.
|
|
60
|
+
*/
|
|
61
|
+
export interface LlmPromptPlugin {
|
|
62
|
+
alias: string
|
|
63
|
+
/** Lower runs first. The built-ins occupy 0 (role), 10 (skills) and 90 (context). */
|
|
64
|
+
order?: number
|
|
65
|
+
/** Contribute static content, before anything has looked at the messages. */
|
|
66
|
+
compose?: (ctx: PromptContext) => void | Promise<void>
|
|
67
|
+
/** Contribute content derived from the messages (detection, lookup, fetch). */
|
|
68
|
+
inspect?: (ctx: PromptContext) => void | Promise<void>
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export interface PromptResult {
|
|
72
|
+
/** The composed system message, or `null` when nothing was contributed. */
|
|
73
|
+
system: MessageFieldWithRole | null
|
|
74
|
+
/** Breakpoints the system prompt consumed; subtract from the message budget. */
|
|
75
|
+
breakpoints: number
|
|
76
|
+
/** The rendered blocks in emission order — exposed for tests and diagnostics. */
|
|
77
|
+
blocks: LlmSystemBlock[]
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export interface PromptServiceOptions {
|
|
81
|
+
/** Skills registered up front; equivalent to calling `register` after construction. */
|
|
82
|
+
skills?: SkillDefinition[]
|
|
83
|
+
/** Plugins registered up front, in addition to the built-ins. */
|
|
84
|
+
plugins?: LlmPromptPlugin[]
|
|
85
|
+
/** Default for `PromptPolicy.cacheSystem`. Defaults to `true`. */
|
|
86
|
+
cacheSystem?: boolean
|
|
87
|
+
/** Default for `PromptPolicy.cacheTtl`. */
|
|
88
|
+
cacheTtl?: CacheTtl
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Registry of reusable skills plus the plugin chain that turns a role and a set of skill
|
|
93
|
+
* aliases into a cache-friendly system prompt.
|
|
94
|
+
*/
|
|
95
|
+
export interface PromptService extends InitializedService {
|
|
96
|
+
/** Register a composition plugin. Runs in `order` then registration order. */
|
|
97
|
+
use: (plugin: LlmPromptPlugin) => void
|
|
98
|
+
/** Register (or replace, by alias) reusable skills. */
|
|
99
|
+
register: (...skills: SkillDefinition[]) => void
|
|
100
|
+
has: (alias: string) => boolean
|
|
101
|
+
/** Resolve aliases to definitions, pulling in `requires` transitively. Unknown aliases are skipped. */
|
|
102
|
+
resolve: (aliases: readonly string[]) => SkillDefinition[]
|
|
103
|
+
/** Every registered skill, in deterministic order. */
|
|
104
|
+
skills: () => SkillDefinition[]
|
|
105
|
+
compose: (
|
|
106
|
+
input: PromptInput,
|
|
107
|
+
messages: readonly MessageFieldWithRole[],
|
|
108
|
+
params: PromptComposeParams,
|
|
109
|
+
) => Promise<PromptResult>
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export interface WithPromptService {
|
|
113
|
+
prompts: () => PromptService
|
|
114
|
+
}
|
package/src/service.ts
CHANGED
|
@@ -51,6 +51,9 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
|
|
|
51
51
|
throw new LlmMissconfiguredError(alias)
|
|
52
52
|
}
|
|
53
53
|
const config: ModelConfig = { ...baseConfig, ...override }
|
|
54
|
+
// The service-wide idle deadline is a floor, not an override: a preset that states its
|
|
55
|
+
// own `streamTimeout` knows something specific about that model and keeps it.
|
|
56
|
+
config.streamTimeout ??= options.streamTimeout
|
|
54
57
|
const preset: Partial<ModelConfig> = config.preset != null
|
|
55
58
|
? { ...(models.find(m => m.alias === config.preset) ?? {}) }
|
|
56
59
|
: {}
|
package/src/types.ts
CHANGED
|
@@ -4,8 +4,10 @@ import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/cor
|
|
|
4
4
|
import type { JSONSchemaType } from 'ajv'
|
|
5
5
|
import type { InitializedService } from '@owlmeans/context'
|
|
6
6
|
import type {
|
|
7
|
-
LlmPurpose, ModelProvider, NullCapture, SpectatorArgument,
|
|
7
|
+
FileProviderRef, LlmPurpose, ModelProvider, NullCapture, SpectatorArgument,
|
|
8
|
+
SpectatorEntryLogged,
|
|
8
9
|
} from '@owlmeans/llm-common'
|
|
10
|
+
import type { PromptInput, PromptService } from './prompt/types.js'
|
|
9
11
|
|
|
10
12
|
export type MaybeArray<T> = T | T[]
|
|
11
13
|
|
|
@@ -47,6 +49,20 @@ export interface LlmLogging {
|
|
|
47
49
|
export interface LlmModelOptions extends LlmLogging {
|
|
48
50
|
model: BaseChatModel
|
|
49
51
|
retries?: number
|
|
52
|
+
/**
|
|
53
|
+
* Role and skills for every call this model makes. Composed into a cacheable system
|
|
54
|
+
* prompt by {@link PromptService}; ignored when no `prompts` resolver is supplied.
|
|
55
|
+
* Usually just `exec.prompt` from the helper execution.
|
|
56
|
+
*/
|
|
57
|
+
prompt?: PromptInput
|
|
58
|
+
/**
|
|
59
|
+
* Late-bound resolver for the prompt service — a function, like `Execution.models`, so
|
|
60
|
+
* the service can be swapped or cloned. Without it the model behaves exactly as before
|
|
61
|
+
* this layer existed: the caller's messages are sent untouched.
|
|
62
|
+
*/
|
|
63
|
+
prompts?: () => PromptService
|
|
64
|
+
/** File access offered to prompt plugins that resolve knowledge from disk. */
|
|
65
|
+
files?: FileProviderRef
|
|
50
66
|
}
|
|
51
67
|
|
|
52
68
|
export type ModelMessage = BaseMessage | MessageFieldWithRole
|
|
@@ -56,10 +72,23 @@ export type ModelInput = MaybeArray<ModelInputItem>
|
|
|
56
72
|
export interface LlmCallOptions {
|
|
57
73
|
/** Short name of the operation — used as the LangChain run name and in spectator entries. */
|
|
58
74
|
action: string
|
|
59
|
-
/**
|
|
75
|
+
/**
|
|
76
|
+
* Cache the leading MESSAGES too. The composed system prompt is cached by default and
|
|
77
|
+
* independently of this flag — this one is about the conversation prefix, which is only
|
|
78
|
+
* worth caching when the same leading messages recur across calls.
|
|
79
|
+
*/
|
|
60
80
|
useCache?: boolean
|
|
61
|
-
/**
|
|
81
|
+
/**
|
|
82
|
+
* How many leading messages form the stable prefix. One breakpoint is placed at its
|
|
83
|
+
* end (not one per message), capped by whatever the system prompt left unspent.
|
|
84
|
+
*/
|
|
62
85
|
cacheMax?: number
|
|
86
|
+
/**
|
|
87
|
+
* Skill aliases for THIS call only. Rendered into the volatile `Context` block, so they
|
|
88
|
+
* never disturb the cached region — declare a skill on the execution instead when it
|
|
89
|
+
* should be part of the shared prefix.
|
|
90
|
+
*/
|
|
91
|
+
skills?: string[]
|
|
63
92
|
}
|
|
64
93
|
|
|
65
94
|
export interface LlmAskOptions extends LlmCallOptions {
|
|
@@ -183,11 +212,33 @@ export interface ModelConfig {
|
|
|
183
212
|
* rotating providers mid-call would flip the structured-output format.
|
|
184
213
|
*/
|
|
185
214
|
fallback?: Partial<ModelConfig>
|
|
215
|
+
/**
|
|
216
|
+
* Per-model override of {@link MIN_CACHEABLE_TOKENS} — the shortest prefix worth a
|
|
217
|
+
* cache breakpoint. Anthropic's own minimum is model-dependent and NOT monotonic
|
|
218
|
+
* across generations (512 on the newest, 1024 on most, 4096 on a few older ones), so a
|
|
219
|
+
* preset that pins an old model should raise this rather than pay for markers that
|
|
220
|
+
* silently never cache.
|
|
221
|
+
*/
|
|
222
|
+
cacheMinTokens?: number
|
|
223
|
+
/**
|
|
224
|
+
* Cache-routing key for providers whose prompt cache is automatic (OpenAI's
|
|
225
|
+
* `prompt_cache_key`): requests sharing a key are routed to the same backend, which
|
|
226
|
+
* raises the hit rate for a shared prefix. Must be STABLE and low-cardinality — one
|
|
227
|
+
* value per role, never per user or per request. Defaults to the config alias.
|
|
228
|
+
*/
|
|
229
|
+
cacheKey?: string
|
|
186
230
|
}
|
|
187
231
|
|
|
188
232
|
export interface LlmServiceOptions {
|
|
189
233
|
/** The full config list; resolved by `alias` on every `getModel` call. */
|
|
190
234
|
models: () => ModelConfig[]
|
|
235
|
+
/**
|
|
236
|
+
* Idle deadline (ms) applied to every model this service builds, unless the model's own
|
|
237
|
+
* config overrides it. This is the knob an application sets where it composes its
|
|
238
|
+
* context — one place to tune how long the whole deployment waits on a silent provider,
|
|
239
|
+
* without touching a preset. Falls back to {@link MODEL_STREAM_TIMEOUT_MS}.
|
|
240
|
+
*/
|
|
241
|
+
streamTimeout?: number
|
|
191
242
|
}
|
|
192
243
|
|
|
193
244
|
/**
|