@owlmeans/llm 0.1.15 → 0.1.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -4
- package/agent-meta/manifest.json +9 -9
- package/agent-meta/skills/llm/SKILL.md +54 -4
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
- package/build/consts.d.ts +36 -3
- package/build/consts.d.ts.map +1 -1
- package/build/consts.js +37 -4
- package/build/consts.js.map +1 -1
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +8 -3
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +20 -1
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts +12 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +22 -0
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/cache.d.ts +17 -0
- package/build/helpers/cache.d.ts.map +1 -0
- package/build/helpers/cache.js +22 -0
- package/build/helpers/cache.js.map +1 -0
- package/build/helpers/index.d.ts +1 -0
- package/build/helpers/index.d.ts.map +1 -1
- package/build/helpers/index.js +1 -0
- package/build/helpers/index.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +7 -0
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +1 -0
- package/build/index.d.ts.map +1 -1
- package/build/index.js +1 -0
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +90 -16
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +136 -21
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +9 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +23 -1
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +38 -5
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/index.d.ts +5 -0
- package/build/prompt/index.d.ts.map +1 -0
- package/build/prompt/index.js +4 -0
- package/build/prompt/index.js.map +1 -0
- package/build/prompt/plugins.d.ts +24 -0
- package/build/prompt/plugins.d.ts.map +1 -0
- package/build/prompt/plugins.js +64 -0
- package/build/prompt/plugins.js.map +1 -0
- package/build/prompt/render.d.ts +28 -0
- package/build/prompt/render.d.ts.map +1 -0
- package/build/prompt/render.js +39 -0
- package/build/prompt/render.js.map +1 -0
- package/build/prompt/service.d.ts +16 -0
- package/build/prompt/service.d.ts.map +1 -0
- package/build/prompt/service.js +145 -0
- package/build/prompt/service.js.map +1 -0
- package/build/prompt/types.d.ts +101 -0
- package/build/prompt/types.d.ts.map +1 -0
- package/build/prompt/types.js +2 -0
- package/build/prompt/types.js.map +1 -0
- package/build/service.d.ts.map +1 -1
- package/build/service.js +3 -0
- package/build/service.js.map +1 -1
- package/build/types.d.ts +53 -3
- package/build/types.d.ts.map +1 -1
- package/build/utils/prompt.d.ts +14 -0
- package/build/utils/prompt.d.ts.map +1 -1
- package/build/utils/prompt.js +32 -0
- package/build/utils/prompt.js.map +1 -1
- package/package.json +13 -6
- package/src/consts.ts +43 -5
- package/src/execution/service.ts +9 -3
- package/src/execution/types.ts +21 -2
- package/src/execution/utils.ts +28 -1
- package/src/helpers/cache.ts +33 -0
- package/src/helpers/index.ts +1 -0
- package/src/helpers/spectate.ts +10 -0
- package/src/index.ts +1 -0
- package/src/model.ts +119 -16
- package/src/plugins/anthropic.ts +159 -20
- package/src/plugins/openai.ts +26 -1
- package/src/plugins/types.ts +42 -5
- package/src/prompt/index.ts +5 -0
- package/src/prompt/plugins.ts +69 -0
- package/src/prompt/render.ts +48 -0
- package/src/prompt/service.ts +194 -0
- package/src/prompt/types.ts +114 -0
- package/src/service.ts +3 -0
- package/src/types.ts +54 -3
- package/src/utils/prompt.ts +33 -0
- package/tests/execution.spec.ts +42 -0
- package/tests/plugins.spec.ts +279 -14
- package/tests/prompt.spec.ts +194 -0
- package/agent-meta/instructions/llm.instructions.md +0 -66
package/src/execution/service.ts
CHANGED
|
@@ -9,7 +9,8 @@ import type {
|
|
|
9
9
|
HelperExecution, TaskExecution, WithExecutionService,
|
|
10
10
|
} from './types.js'
|
|
11
11
|
import {
|
|
12
|
-
composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
|
|
12
|
+
composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
|
|
13
|
+
mergePrompt, resolveRole,
|
|
13
14
|
} from './utils.js'
|
|
14
15
|
|
|
15
16
|
/**
|
|
@@ -50,18 +51,21 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
50
51
|
level: ExecutionLevel.Project,
|
|
51
52
|
purpose: { ...input.purpose },
|
|
52
53
|
policy: { ...input.policy },
|
|
54
|
+
...(input.prompt != null ? { prompt: { ...input.prompt } } : {}),
|
|
53
55
|
}) as S['project'],
|
|
54
56
|
|
|
55
57
|
forTask: (parent, input) => {
|
|
56
|
-
const { effort, phase, data, ...extras } = input
|
|
58
|
+
const { effort, phase, data, prompt, ...extras } = input
|
|
57
59
|
const policy = effort != null
|
|
58
60
|
? mergePolicy(parent.policy, { effort })
|
|
59
61
|
: { ...parent.policy }
|
|
62
|
+
const merged = mergePrompt(parent.prompt, prompt)
|
|
60
63
|
|
|
61
64
|
// Spreading the parent carries every collaborator and domain field forward; the
|
|
62
65
|
// task's own state is composed afterwards, from the seeded resumable fields.
|
|
63
66
|
const taskExec = {
|
|
64
67
|
...parent, ...extras, level: ExecutionLevel.Task, purpose: { ...parent.purpose }, policy,
|
|
68
|
+
...(merged != null ? { prompt: merged } : {}),
|
|
65
69
|
} as unknown as TaskExecution
|
|
66
70
|
;(taskExec as { state: TaskExecutionState }).state = composeTaskState({
|
|
67
71
|
...taskExec,
|
|
@@ -78,9 +82,10 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
78
82
|
},
|
|
79
83
|
|
|
80
84
|
forHelper: (parent, input) => {
|
|
81
|
-
const { role, effort, dedication, ...extras } = input
|
|
85
|
+
const { role, effort, dedication, prompt, ...extras } = input
|
|
82
86
|
const localPolicy = effort != null ? mergePolicy(parent.policy, { effort }) : parent.policy
|
|
83
87
|
const scoped = { ...parent, policy: localPolicy } as S['exec']
|
|
88
|
+
const merged = mergePrompt(parent.prompt, prompt)
|
|
84
89
|
|
|
85
90
|
const helperExec = {
|
|
86
91
|
...parent,
|
|
@@ -90,6 +95,7 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
90
95
|
? { ...parent.purpose, dedication }
|
|
91
96
|
: { ...parent.purpose },
|
|
92
97
|
policy: localPolicy,
|
|
98
|
+
...(merged != null ? { prompt: merged } : {}),
|
|
93
99
|
role: resolveRole(localPolicy, role),
|
|
94
100
|
model: self().model(scoped, role),
|
|
95
101
|
temperatureFactory: self().temperatureFactory(scoped, role),
|
package/src/execution/types.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
2
|
import type { InitializedService } from '@owlmeans/context'
|
|
3
3
|
import type {
|
|
4
|
-
ExecutionEffort, ExecutionLevel, ExecutionState,
|
|
5
|
-
ModelPolicy, ModelRole, TaskExecutionState,
|
|
4
|
+
ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef, LlmPurpose,
|
|
5
|
+
ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
|
|
6
6
|
} from '@owlmeans/llm-common'
|
|
7
7
|
import type { LlmService, TemperatureFactory } from '../types.js'
|
|
8
|
+
import type { PromptService } from '../prompt/types.js'
|
|
8
9
|
|
|
9
10
|
/**
|
|
10
11
|
* Runtime execution = serializable {@link ExecutionState} + attached collaborators.
|
|
@@ -18,6 +19,10 @@ import type { LlmService, TemperatureFactory } from '../types.js'
|
|
|
18
19
|
export interface Execution extends ExecutionState {
|
|
19
20
|
/** Resolver for the model factory — a function so the service can be swapped/cloned. */
|
|
20
21
|
models: () => LlmService
|
|
22
|
+
/** Resolver for the skill registry / prompt composer. Same late-binding rationale. */
|
|
23
|
+
prompts?: () => PromptService
|
|
24
|
+
/** File access offered to prompt plugins. Declared a collaborator, never snapshotted. */
|
|
25
|
+
files?: FileProviderRef
|
|
21
26
|
outputErrors?: boolean
|
|
22
27
|
captureNull?: boolean
|
|
23
28
|
}
|
|
@@ -42,8 +47,12 @@ export interface HelperExecution extends Execution {
|
|
|
42
47
|
|
|
43
48
|
export interface ProjectExecutionInput {
|
|
44
49
|
models: () => LlmService
|
|
50
|
+
prompts?: () => PromptService
|
|
51
|
+
files?: FileProviderRef
|
|
45
52
|
policy: ModelPolicy
|
|
46
53
|
purpose: LlmPurpose
|
|
54
|
+
/** Baseline role and skills for the whole run. */
|
|
55
|
+
prompt?: PromptPolicy
|
|
47
56
|
outputErrors?: boolean
|
|
48
57
|
captureNull?: boolean
|
|
49
58
|
}
|
|
@@ -51,15 +60,23 @@ export interface ProjectExecutionInput {
|
|
|
51
60
|
export interface TaskExecutionInput {
|
|
52
61
|
/** Raise (or lower) the effort tier for this task and everything derived from it. */
|
|
53
62
|
effort?: ExecutionEffort
|
|
63
|
+
/** Skills (and optionally a role) layered on top of the project's. Skills accumulate. */
|
|
64
|
+
prompt?: PromptPolicy
|
|
54
65
|
/** Optional seeds for the resumable task state. */
|
|
55
66
|
phase?: string
|
|
56
67
|
data?: Record<string, unknown>
|
|
57
68
|
}
|
|
58
69
|
|
|
59
70
|
export interface HelperExecutionInput {
|
|
71
|
+
/**
|
|
72
|
+
* Which MODEL to use. Distinct from `prompt.role`, which is the system-prompt text
|
|
73
|
+
* defining the persona — one selects hardware, the other writes the job description.
|
|
74
|
+
*/
|
|
60
75
|
role: ModelRole
|
|
61
76
|
/** Local effort bump without escalating the whole branch. */
|
|
62
77
|
effort?: ExecutionEffort
|
|
78
|
+
/** The helper's persona and its own skills, layered on top of the task's. */
|
|
79
|
+
prompt?: PromptPolicy
|
|
63
80
|
/** Refines `purpose.dedication`. */
|
|
64
81
|
dedication?: string
|
|
65
82
|
}
|
|
@@ -67,6 +84,8 @@ export interface HelperExecutionInput {
|
|
|
67
84
|
/** Collaborators re-attached to a state that was restored from storage. */
|
|
68
85
|
export interface RestoreCollaborators {
|
|
69
86
|
models?: () => LlmService
|
|
87
|
+
prompts?: () => PromptService
|
|
88
|
+
files?: FileProviderRef
|
|
70
89
|
}
|
|
71
90
|
|
|
72
91
|
/**
|
package/src/execution/utils.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { ExecutionLevel } from '@owlmeans/llm-common'
|
|
2
2
|
import type {
|
|
3
3
|
ExecutionEffort, ExecutionState, ModelConfigOverride, ModelConfigPatch,
|
|
4
|
-
ModelPolicy, ModelRole, TaskExecutionState,
|
|
4
|
+
ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
|
|
5
5
|
} from '@owlmeans/llm-common'
|
|
6
6
|
import { EFFORT_TABLE } from '../consts.js'
|
|
7
7
|
import type { Execution, TaskExecution } from './types.js'
|
|
@@ -19,6 +19,33 @@ export const mergePolicy = (base: ModelPolicy, patch: Partial<ModelPolicy>): Mod
|
|
|
19
19
|
: undefined,
|
|
20
20
|
})
|
|
21
21
|
|
|
22
|
+
/**
|
|
23
|
+
* Overlay a prompt policy onto the one inherited from the parent level.
|
|
24
|
+
*
|
|
25
|
+
* Skills ACCUMULATE — a task adds to what the project declared, a helper adds to the
|
|
26
|
+
* task — because that is how a capability set is built up as work narrows. The role is
|
|
27
|
+
* replaced instead: the deepest level that names one owns the persona.
|
|
28
|
+
*
|
|
29
|
+
* The union preserves first-seen order and de-duplicates, so the composed prompt is
|
|
30
|
+
* byte-identical no matter how many levels contributed the same skill.
|
|
31
|
+
*/
|
|
32
|
+
export const mergePrompt = (
|
|
33
|
+
base: PromptPolicy | undefined,
|
|
34
|
+
patch: PromptPolicy | undefined,
|
|
35
|
+
): PromptPolicy | undefined => {
|
|
36
|
+
if (base == null && patch == null) {
|
|
37
|
+
return undefined
|
|
38
|
+
}
|
|
39
|
+
const skills = [...new Set([...(base?.skills ?? []), ...(patch?.skills ?? [])])]
|
|
40
|
+
|
|
41
|
+
return {
|
|
42
|
+
...base,
|
|
43
|
+
...patch,
|
|
44
|
+
...(base?.role != null || patch?.role != null ? { role: patch?.role ?? base?.role } : {}),
|
|
45
|
+
...(skills.length > 0 ? { skills } : {}),
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
22
49
|
/** Apply the policy's role→role remap. */
|
|
23
50
|
export const resolveRole = (policy: ModelPolicy, role: ModelRole): ModelRole =>
|
|
24
51
|
(policy.roleOverrides?.[role] as ModelRole | undefined) ?? role
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import type { UsageMetadata } from '@langchain/core/messages'
|
|
2
|
+
import type { CacheUsage } from '@owlmeans/llm-common'
|
|
3
|
+
|
|
4
|
+
/** LangChain normalizes every provider's cache accounting into these two fields. */
|
|
5
|
+
interface InputTokenDetails {
|
|
6
|
+
cache_read?: number
|
|
7
|
+
cache_creation?: number
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Prompt-cache accounting for one completion.
|
|
12
|
+
*
|
|
13
|
+
* This is the only honest answer to "is caching actually working". A composed prefix can
|
|
14
|
+
* look perfectly stable and still miss on every call — a stray timestamp, a set iterated
|
|
15
|
+
* in a different order, a tool list rebuilt per request. If `read` stays at zero across
|
|
16
|
+
* repeated calls that share a prefix, something is invalidating it; diff the rendered
|
|
17
|
+
* blocks between two calls to find out what.
|
|
18
|
+
*/
|
|
19
|
+
export const readCacheUsage = (message: { usage_metadata?: UsageMetadata }): CacheUsage => {
|
|
20
|
+
const usage = message.usage_metadata
|
|
21
|
+
const details = usage?.input_token_details as InputTokenDetails | undefined
|
|
22
|
+
|
|
23
|
+
return {
|
|
24
|
+
read: details?.cache_read ?? 0,
|
|
25
|
+
creation: details?.cache_creation ?? 0,
|
|
26
|
+
input: usage?.input_tokens ?? 0,
|
|
27
|
+
output: usage?.output_tokens ?? 0,
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** `true` when the provider reported any cache activity at all. */
|
|
32
|
+
export const hasCacheActivity = (usage: CacheUsage): boolean =>
|
|
33
|
+
usage.read > 0 || usage.creation > 0
|
package/src/helpers/index.ts
CHANGED
package/src/helpers/spectate.ts
CHANGED
|
@@ -3,6 +3,7 @@ import type { UsageMetadata } from '@langchain/core/messages'
|
|
|
3
3
|
import { SpectatorContentType } from '@owlmeans/llm-common'
|
|
4
4
|
import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
|
|
5
5
|
import type { LlmSpectator, ModelInputItem } from '../types.js'
|
|
6
|
+
import { hasCacheActivity, readCacheUsage } from './cache.js'
|
|
6
7
|
|
|
7
8
|
/** Normalize one prompt message into the spectator's storage shape. */
|
|
8
9
|
const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
|
|
@@ -63,5 +64,14 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
|
|
|
63
64
|
completion.content = message.content as unknown as string
|
|
64
65
|
}
|
|
65
66
|
|
|
67
|
+
// Silent unless the provider reported cache activity, so it costs nothing when
|
|
68
|
+
// caching is off — and is the one signal that tells a stable prefix from a broken one.
|
|
69
|
+
const cache = readCacheUsage(message)
|
|
70
|
+
if (hasCacheActivity(cache)) {
|
|
71
|
+
console.log(
|
|
72
|
+
`Prompt cache [${action}]: read ${cache.read}, written ${cache.creation}, uncached ${cache.input}`
|
|
73
|
+
)
|
|
74
|
+
}
|
|
75
|
+
|
|
66
76
|
return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
|
|
67
77
|
}
|
package/src/index.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './model.js'
|
|
|
6
6
|
export * from './service.js'
|
|
7
7
|
export * from './helpers/index.js'
|
|
8
8
|
export * from './execution/index.js'
|
|
9
|
+
export * from './prompt/index.js'
|
|
9
10
|
export type * from './plugins/types.js'
|
|
10
11
|
export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
|
|
11
12
|
export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
|
package/src/model.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { Ajv } from 'ajv'
|
|
2
2
|
import type { JSONSchemaType } from 'ajv'
|
|
3
|
-
import { AIMessage } from '@langchain/core/messages'
|
|
4
|
-
import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
|
|
3
|
+
import { AIMessage, BaseMessage } from '@langchain/core/messages'
|
|
4
|
+
import type { AIMessageChunk, MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
|
|
5
5
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
6
6
|
import { StructuredMode } from '@owlmeans/llm-common'
|
|
7
7
|
import type { NullKind } from '@owlmeans/llm-common'
|
|
8
8
|
import {
|
|
9
|
-
DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
|
|
9
|
+
DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
|
|
10
10
|
} from './consts.js'
|
|
11
11
|
import { LlmModelError } from './errors.js'
|
|
12
12
|
import { pluginFor, pluginOf } from './plugins/index.js'
|
|
@@ -18,7 +18,7 @@ import { spectate } from './helpers/spectate.js'
|
|
|
18
18
|
import { idleTimeout, readConfig } from './utils/config.js'
|
|
19
19
|
import { reportNull } from './utils/null-report.js'
|
|
20
20
|
import type { NullReportParams } from './utils/null-report.js'
|
|
21
|
-
import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
|
|
21
|
+
import { applyNoThink, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
|
|
22
22
|
import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
|
|
23
23
|
import { streamWithDeadline } from './utils/stream.js'
|
|
24
24
|
import type {
|
|
@@ -28,6 +28,48 @@ import type {
|
|
|
28
28
|
|
|
29
29
|
type StreamOptions = Parameters<BaseChatModel['stream']>[1]
|
|
30
30
|
|
|
31
|
+
const isSystem = (msg: MessageFieldWithRole): boolean =>
|
|
32
|
+
msg instanceof BaseMessage ? msg.getType() === 'system' : `${msg.role}` === 'system'
|
|
33
|
+
|
|
34
|
+
const textOf = (content: MessageContent | undefined): string => {
|
|
35
|
+
if (typeof content === 'string') {
|
|
36
|
+
return content.trim()
|
|
37
|
+
}
|
|
38
|
+
if (Array.isArray(content)) {
|
|
39
|
+
return content
|
|
40
|
+
.map(part => {
|
|
41
|
+
const text = (part as unknown as { text?: unknown }).text
|
|
42
|
+
return typeof text === 'string' ? text : ''
|
|
43
|
+
})
|
|
44
|
+
.filter(text => text !== '')
|
|
45
|
+
.join('\n\n')
|
|
46
|
+
.trim()
|
|
47
|
+
}
|
|
48
|
+
return ''
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Detach the caller's LEADING system messages and return their text.
|
|
53
|
+
*
|
|
54
|
+
* They are re-emitted as the `Context` block of the composed prompt, which is what keeps
|
|
55
|
+
* a caller that still builds its own `SystemMessage` working unchanged — the text simply
|
|
56
|
+
* travels a different route and lands in the same place. Only the leading run is taken:
|
|
57
|
+
* a system message deliberately placed mid-conversation is an operator instruction whose
|
|
58
|
+
* position carries meaning, and moving it would change what the model sees.
|
|
59
|
+
*/
|
|
60
|
+
const takeLeadingSystem = (msgs: MessageFieldWithRole[]): string[] => {
|
|
61
|
+
const carried: string[] = []
|
|
62
|
+
while (msgs.length > 0 && isSystem(msgs[0])) {
|
|
63
|
+
const [msg] = msgs.splice(0, 1)
|
|
64
|
+
const text = textOf(msg.content)
|
|
65
|
+
if (text !== '') {
|
|
66
|
+
carried.push(text)
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return carried
|
|
71
|
+
}
|
|
72
|
+
|
|
31
73
|
/**
|
|
32
74
|
* Build the four-method model API on top of a LangChain chat model.
|
|
33
75
|
*
|
|
@@ -43,6 +85,9 @@ export const makeLlmModel = ({
|
|
|
43
85
|
captureNull = false,
|
|
44
86
|
retries = DEFAULT_MODEL_RETRIES,
|
|
45
87
|
purpose,
|
|
88
|
+
prompt,
|
|
89
|
+
prompts,
|
|
90
|
+
files,
|
|
46
91
|
}: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
|
|
47
92
|
|
|
48
93
|
const ajv = new Ajv({ strict: false })
|
|
@@ -53,15 +98,67 @@ export const makeLlmModel = ({
|
|
|
53
98
|
const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
|
|
54
99
|
const timeout = idleTimeout(config)
|
|
55
100
|
|
|
56
|
-
/**
|
|
57
|
-
|
|
101
|
+
/**
|
|
102
|
+
* Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
|
|
103
|
+
* dependency order.
|
|
104
|
+
*
|
|
105
|
+
* With no prompt service wired this is exactly what it always was — the caller's
|
|
106
|
+
* messages, a JSON nudge, `/no_think`, cache markers. With one, the caller's leading
|
|
107
|
+
* system text is folded into a composed prompt whose stable sections come first, which
|
|
108
|
+
* is the whole point: a prompt cache is a PREFIX match, so the bytes every call shares
|
|
109
|
+
* have to be physically ahead of the bytes that differ.
|
|
110
|
+
*/
|
|
111
|
+
const prepare = async (
|
|
112
|
+
input: ModelInput,
|
|
113
|
+
action: string,
|
|
114
|
+
useCache: boolean,
|
|
115
|
+
cacheMax: number,
|
|
116
|
+
json: boolean,
|
|
117
|
+
callSkills?: string[],
|
|
118
|
+
): Promise<MessageFieldWithRole[]> => {
|
|
58
119
|
const msgs = normalizeInput(input)
|
|
120
|
+
// The caller may hand back messages this pipeline marked on a PREVIOUS call — the
|
|
121
|
+
// markers live on its own objects. The budget is per request, so clear them and
|
|
122
|
+
// re-place our own below; otherwise they accumulate until the provider 400s.
|
|
123
|
+
stripCacheMarkers(msgs)
|
|
124
|
+
let reserved = 0
|
|
125
|
+
|
|
126
|
+
if (prompts != null) {
|
|
127
|
+
const carried = takeLeadingSystem(msgs)
|
|
128
|
+
const composed = await prompts().compose(
|
|
129
|
+
{
|
|
130
|
+
...prompt,
|
|
131
|
+
context: [...(prompt?.context ?? []), ...carried],
|
|
132
|
+
callSkills: callSkills ?? prompt?.callSkills,
|
|
133
|
+
},
|
|
134
|
+
msgs,
|
|
135
|
+
{ model, provider: plugin, purpose, action, cacheMax, files },
|
|
136
|
+
)
|
|
137
|
+
if (composed.system != null) {
|
|
138
|
+
msgs.unshift(composed.system)
|
|
139
|
+
reserved = composed.breakpoints
|
|
140
|
+
} else {
|
|
141
|
+
// Defensive: nothing was contributed, so hand the caller's own text straight back.
|
|
142
|
+
for (let i = carried.length - 1; i >= 0; i--) {
|
|
143
|
+
msgs.unshift({ role: 'system', content: carried[i] })
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
59
148
|
if (json) ensureJsonMention(msgs)
|
|
60
149
|
applyNoThink(msgs, config.disableThinking)
|
|
61
150
|
// Cache markers replace string content with content blocks, so they must go last.
|
|
62
|
-
|
|
63
|
-
|
|
151
|
+
const ttl = prompt?.cacheTtl
|
|
152
|
+
const marked = plugin?.patchCache?.(msgs, {
|
|
153
|
+
model, useCache, cacheMax, reserved, ...(ttl != null ? { ttl } : {}),
|
|
154
|
+
})
|
|
155
|
+
if (reserved > 0 || marked === true) {
|
|
156
|
+
console.log(
|
|
157
|
+
`Prompt caching for ${plugin?.type}: ${reserved} system breakpoint(s)`
|
|
158
|
+
+ `${marked === true ? ', 1 message breakpoint' : ''}`
|
|
159
|
+
)
|
|
64
160
|
}
|
|
161
|
+
|
|
65
162
|
return msgs
|
|
66
163
|
}
|
|
67
164
|
|
|
@@ -197,8 +294,11 @@ export const makeLlmModel = ({
|
|
|
197
294
|
}
|
|
198
295
|
|
|
199
296
|
const helper: LlmModel = {
|
|
200
|
-
ask: async (
|
|
201
|
-
|
|
297
|
+
ask: async (
|
|
298
|
+
input,
|
|
299
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmAskOptions
|
|
300
|
+
) => {
|
|
301
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
202
302
|
return withRetry({ retries, outputErrors }, async i => {
|
|
203
303
|
const refined = refineModel(i)
|
|
204
304
|
console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
|
|
@@ -242,8 +342,11 @@ export const makeLlmModel = ({
|
|
|
242
342
|
})
|
|
243
343
|
},
|
|
244
344
|
|
|
245
|
-
talk: async (
|
|
246
|
-
|
|
345
|
+
talk: async (
|
|
346
|
+
input,
|
|
347
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmTalkOptions
|
|
348
|
+
) => {
|
|
349
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
247
350
|
return withRetry({ retries, outputErrors }, async i => {
|
|
248
351
|
const refined = refineModel(i)
|
|
249
352
|
console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
|
|
@@ -277,9 +380,9 @@ export const makeLlmModel = ({
|
|
|
277
380
|
invoke: async <T>(
|
|
278
381
|
input: ModelInput,
|
|
279
382
|
schema: JSONSchemaType<T>,
|
|
280
|
-
{ temperature, ref, filter, action, useCache = false, cacheMax =
|
|
383
|
+
{ temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmInvokeOptions<T>
|
|
281
384
|
) => {
|
|
282
|
-
const msgs = prepare(input, useCache, cacheMax, true)
|
|
385
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
283
386
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
284
387
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
285
388
|
|
|
@@ -325,9 +428,9 @@ export const makeLlmModel = ({
|
|
|
325
428
|
request: async <T>(
|
|
326
429
|
input: ModelInput,
|
|
327
430
|
schema: JSONSchemaType<T>,
|
|
328
|
-
{ ref, filter, action, useCache = false, cacheMax =
|
|
431
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmRequestOptions
|
|
329
432
|
) => {
|
|
330
|
-
const msgs = prepare(input, useCache, cacheMax, true)
|
|
433
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
331
434
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
332
435
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
333
436
|
|
package/src/plugins/anthropic.ts
CHANGED
|
@@ -1,9 +1,12 @@
|
|
|
1
1
|
import { ChatAnthropic } from '@langchain/anthropic'
|
|
2
2
|
import { BadRequestError } from '@anthropic-ai/sdk'
|
|
3
3
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
4
|
-
import {
|
|
4
|
+
import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
|
|
5
|
+
import { ModelProvider, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
|
|
6
|
+
import type { CacheTtl } from '@owlmeans/llm-common'
|
|
5
7
|
import type { LlmPlugin } from './types.js'
|
|
6
|
-
import { MAX_CACHE_BREAKPOINTS } from '../consts.js'
|
|
8
|
+
import { CHARS_PER_TOKEN, MAX_CACHE_BREAKPOINTS, MIN_CACHEABLE_TOKENS } from '../consts.js'
|
|
9
|
+
import { readConfig } from '../utils/config.js'
|
|
7
10
|
import { escalateMaxTokens, makeClientOptions } from './utils.js'
|
|
8
11
|
|
|
9
12
|
/** Model-name prefix that supports prompt caching through `cache_control` markers. */
|
|
@@ -11,6 +14,64 @@ const CACHEABLE_PREFIX = 'claude-'
|
|
|
11
14
|
|
|
12
15
|
export const ANTHROPIC_FAMILY = 'anthropic'
|
|
13
16
|
|
|
17
|
+
type ContentBlock = Record<string, unknown>
|
|
18
|
+
|
|
19
|
+
const supportsCache = (model: BaseChatModel): boolean =>
|
|
20
|
+
(model as ChatAnthropic).modelName?.startsWith(CACHEABLE_PREFIX) === true
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Shortest prefix worth a breakpoint, in characters. Anthropic silently declines to
|
|
24
|
+
* create an entry below its own per-model minimum, so a marker there wastes one of the
|
|
25
|
+
* four breakpoints and reports a cache that was never written.
|
|
26
|
+
*/
|
|
27
|
+
const minCacheableChars = (model: BaseChatModel): number =>
|
|
28
|
+
(readConfig(model).cacheMinTokens ?? MIN_CACHEABLE_TOKENS) * CHARS_PER_TOKEN
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* The marker itself. `ttl` is omitted for the 5-minute default so the emitted bytes stay
|
|
32
|
+
* the classic shape — a request that differs only in an explicit `"ttl": "5m"` would not
|
|
33
|
+
* match a prefix cached without it.
|
|
34
|
+
*/
|
|
35
|
+
const marker = (ttl: CacheTtl): ContentBlock =>
|
|
36
|
+
ttl === '1h' ? { type: 'ephemeral', ttl: '1h' } : { type: 'ephemeral' }
|
|
37
|
+
|
|
38
|
+
const contentLength = (content: MessageContent | undefined): number => {
|
|
39
|
+
if (typeof content === 'string') {
|
|
40
|
+
return content.length
|
|
41
|
+
}
|
|
42
|
+
if (Array.isArray(content)) {
|
|
43
|
+
return content.reduce<number>((sum, part) => {
|
|
44
|
+
const text = (part as unknown as { text?: unknown }).text
|
|
45
|
+
return sum + (typeof text === 'string' ? text.length : 0)
|
|
46
|
+
}, 0)
|
|
47
|
+
}
|
|
48
|
+
return 0
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Put a breakpoint on a message's LAST content block, lifting string content into a block
|
|
53
|
+
* so the marker has somewhere to live. Idempotent, and it never mutates a block the
|
|
54
|
+
* caller owns — the array is rebuilt around a fresh copy of the final entry.
|
|
55
|
+
*/
|
|
56
|
+
const markMessage = (msg: MessageFieldWithRole, mark: ContentBlock): boolean => {
|
|
57
|
+
if (typeof msg.content === 'string') {
|
|
58
|
+
msg.content = [{ type: 'text', text: msg.content, cache_control: mark }] as unknown as MessageContent
|
|
59
|
+
return true
|
|
60
|
+
}
|
|
61
|
+
if (Array.isArray(msg.content) && msg.content.length > 0) {
|
|
62
|
+
const blocks = [...msg.content] as ContentBlock[]
|
|
63
|
+
const last = blocks[blocks.length - 1]
|
|
64
|
+
if (last.cache_control != null) {
|
|
65
|
+
return true
|
|
66
|
+
}
|
|
67
|
+
blocks[blocks.length - 1] = { ...last, cache_control: mark }
|
|
68
|
+
msg.content = blocks as unknown as MessageContent
|
|
69
|
+
return true
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
return false
|
|
73
|
+
}
|
|
74
|
+
|
|
14
75
|
export const anthropicPlugin: LlmPlugin = {
|
|
15
76
|
type: ModelProvider.Anthropic,
|
|
16
77
|
|
|
@@ -70,28 +131,106 @@ export const anthropicPlugin: LlmPlugin = {
|
|
|
70
131
|
},
|
|
71
132
|
|
|
72
133
|
/**
|
|
73
|
-
*
|
|
74
|
-
*
|
|
75
|
-
*
|
|
134
|
+
* Render the composed system prompt as Anthropic content blocks, one per section, with
|
|
135
|
+
* a breakpoint on each stability boundary:
|
|
136
|
+
*
|
|
137
|
+
* - after `Role` + `Skills` — the region every call of this role shares;
|
|
138
|
+
* - after `Packages` — varies with what the request mentions, so it gets its own
|
|
139
|
+
* entry and can never invalidate the block above it;
|
|
140
|
+
* - after the last block — so the whole system prompt is cached, which is the
|
|
141
|
+
* default this layer promises.
|
|
142
|
+
*
|
|
143
|
+
* Boundaries that coincide collapse into one. Marking stops as soon as the budget is
|
|
144
|
+
* spent, earliest boundary first — the earliest prefix is the one most calls share.
|
|
76
145
|
*/
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
for (const msg of msgs) {
|
|
82
|
-
msg.content = typeof msg.content === 'string' ? [{
|
|
83
|
-
type: 'text',
|
|
84
|
-
text: msg.content,
|
|
85
|
-
cache_control: { type: 'ephemeral' },
|
|
86
|
-
}] : msg.content
|
|
87
|
-
if (++i > max - 1) break
|
|
146
|
+
patchSystem: (blocks, { model, cacheMax, ttl }) => {
|
|
147
|
+
const content: ContentBlock[] = blocks.map(block => ({ type: 'text', text: block.text }))
|
|
148
|
+
if (!supportsCache(model) || cacheMax < 1) {
|
|
149
|
+
return { content: content as unknown as MessageContent, breakpoints: 0 }
|
|
88
150
|
}
|
|
89
|
-
|
|
151
|
+
|
|
152
|
+
const lastOf = (...wanted: PromptBlock[]): number =>
|
|
153
|
+
blocks.reduce((found, block, i) => wanted.includes(block.block) ? i : found, -1)
|
|
154
|
+
|
|
155
|
+
// Closing the prompt is worth a breakpoint only when the last block is STABLE. A
|
|
156
|
+
// trailing `Context` changes every call, so marking it would pay a cache write every
|
|
157
|
+
// time and never read one back — it burns a breakpoint to buy nothing. A prompt that
|
|
158
|
+
// is ONLY context (a caller that has not adopted role/skills) is still worth marking,
|
|
159
|
+
// because there it IS the stable part.
|
|
160
|
+
const last = blocks.length - 1
|
|
161
|
+
const closing = blocks[last].block === PromptBlock.Context && blocks.length > 1 ? -1 : last
|
|
162
|
+
|
|
163
|
+
const boundaries = [...new Set([
|
|
164
|
+
lastOf(PromptBlock.Role, PromptBlock.Skills),
|
|
165
|
+
lastOf(PromptBlock.Packages),
|
|
166
|
+
closing,
|
|
167
|
+
].filter(index => index >= 0))].sort((a, b) => a - b)
|
|
168
|
+
|
|
169
|
+
const minChars = minCacheableChars(model)
|
|
170
|
+
let consumed = 0
|
|
171
|
+
let chars = 0
|
|
172
|
+
let next = 0
|
|
173
|
+
for (let i = 0; i < content.length; i++) {
|
|
174
|
+
chars += blocks[i].text.length
|
|
175
|
+
if (i !== boundaries[next]) {
|
|
176
|
+
continue
|
|
177
|
+
}
|
|
178
|
+
next++
|
|
179
|
+
if (chars < minChars || consumed >= cacheMax) {
|
|
180
|
+
continue
|
|
181
|
+
}
|
|
182
|
+
content[i].cache_control = marker(ttl)
|
|
183
|
+
consumed++
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
return { content: content as unknown as MessageContent, breakpoints: consumed }
|
|
187
|
+
},
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* One breakpoint, at the end of the stable message prefix (`cacheMax` messages).
|
|
191
|
+
*
|
|
192
|
+
* Not one marker per message: the request budget is {@link MAX_CACHE_BREAKPOINTS} in
|
|
193
|
+
* total across tools, system and messages, and the system prompt — the part that is
|
|
194
|
+
* genuinely identical between calls — has first claim on it. `reserved` is what the
|
|
195
|
+
* system prompt already spent.
|
|
196
|
+
*/
|
|
197
|
+
patchCache: (msgs, { model, useCache, cacheMax, reserved = 0, ttl = '5m' }) => {
|
|
198
|
+
if (!useCache || !supportsCache(model) || msgs.length === 0) {
|
|
199
|
+
return false
|
|
200
|
+
}
|
|
201
|
+
if (Math.min(cacheMax, MAX_CACHE_BREAKPOINTS - reserved) < 1) {
|
|
202
|
+
return false
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// The final message is the per-call payload, and `ensureJsonMention` / `applyNoThink`
|
|
206
|
+
// append to it — including it in the prefix would write a fresh entry every call and
|
|
207
|
+
// read none. The stable prefix therefore stops one short of the end.
|
|
208
|
+
const index = Math.min(cacheMax, msgs.length - 1) - 1
|
|
209
|
+
const target = index >= 0 ? msgs[index] : null
|
|
210
|
+
if (target == null) {
|
|
211
|
+
return false
|
|
212
|
+
}
|
|
213
|
+
const chars = msgs.slice(0, index + 1)
|
|
214
|
+
.reduce((sum, msg) => sum + contentLength(msg.content), 0)
|
|
215
|
+
if (chars < minCacheableChars(model)) {
|
|
216
|
+
return false
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
return markMessage(target, marker(ttl))
|
|
90
220
|
},
|
|
91
221
|
|
|
92
222
|
/**
|
|
93
|
-
* A malformed request (bad schema, unsupported parameter, oversized `max_tokens
|
|
94
|
-
* cannot be fixed by retrying — surface it immediately instead
|
|
223
|
+
* A malformed request (bad schema, unsupported parameter, oversized `max_tokens`, too
|
|
224
|
+
* many cache breakpoints) cannot be fixed by retrying — surface it immediately instead
|
|
225
|
+
* of burning the budget.
|
|
226
|
+
*
|
|
227
|
+
* The `status` check is not redundant with the `instanceof`: `@langchain/anthropic`
|
|
228
|
+
* carries its OWN nested copy of `@anthropic-ai/sdk`, so the error it throws is an
|
|
229
|
+
* instance of a DIFFERENT `BadRequestError` class than the one imported here and the
|
|
230
|
+
* `instanceof` silently fails. That turned every fatal 400 into eight full retries —
|
|
231
|
+
* a single malformed request became minutes of thrash with the real cause buried.
|
|
95
232
|
*/
|
|
96
|
-
isFatal: e => e instanceof BadRequestError
|
|
233
|
+
isFatal: e => e instanceof BadRequestError || (e as { status?: unknown })?.status === 400
|
|
234
|
+
? e as Error
|
|
235
|
+
: null,
|
|
97
236
|
}
|