@owlmeans/llm 0.1.14 → 0.1.16-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agent-meta/instructions/llm-prompt-caching.instructions.md +100 -0
- package/agent-meta/instructions/llm.instructions.md +13 -3
- package/agent-meta/manifest.json +16 -2
- package/agent-meta/skills/llm/SKILL.md +54 -4
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
- package/build/consts.d.ts +36 -3
- package/build/consts.d.ts.map +1 -1
- package/build/consts.js +37 -4
- package/build/consts.js.map +1 -1
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +8 -3
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +20 -1
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts +12 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +22 -0
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/cache.d.ts +17 -0
- package/build/helpers/cache.d.ts.map +1 -0
- package/build/helpers/cache.js +22 -0
- package/build/helpers/cache.js.map +1 -0
- package/build/helpers/index.d.ts +1 -0
- package/build/helpers/index.d.ts.map +1 -1
- package/build/helpers/index.js +1 -0
- package/build/helpers/index.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +7 -0
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +1 -0
- package/build/index.d.ts.map +1 -1
- package/build/index.js +1 -0
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +90 -16
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +136 -21
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +9 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +23 -1
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +38 -5
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/index.d.ts +5 -0
- package/build/prompt/index.d.ts.map +1 -0
- package/build/prompt/index.js +4 -0
- package/build/prompt/index.js.map +1 -0
- package/build/prompt/plugins.d.ts +24 -0
- package/build/prompt/plugins.d.ts.map +1 -0
- package/build/prompt/plugins.js +64 -0
- package/build/prompt/plugins.js.map +1 -0
- package/build/prompt/render.d.ts +28 -0
- package/build/prompt/render.d.ts.map +1 -0
- package/build/prompt/render.js +39 -0
- package/build/prompt/render.js.map +1 -0
- package/build/prompt/service.d.ts +16 -0
- package/build/prompt/service.d.ts.map +1 -0
- package/build/prompt/service.js +145 -0
- package/build/prompt/service.js.map +1 -0
- package/build/prompt/types.d.ts +101 -0
- package/build/prompt/types.d.ts.map +1 -0
- package/build/prompt/types.js +2 -0
- package/build/prompt/types.js.map +1 -0
- package/build/service.d.ts.map +1 -1
- package/build/service.js +3 -0
- package/build/service.js.map +1 -1
- package/build/types.d.ts +53 -3
- package/build/types.d.ts.map +1 -1
- package/build/utils/prompt.d.ts +14 -0
- package/build/utils/prompt.d.ts.map +1 -1
- package/build/utils/prompt.js +32 -0
- package/build/utils/prompt.js.map +1 -1
- package/package.json +13 -6
- package/src/consts.ts +43 -5
- package/src/execution/service.ts +9 -3
- package/src/execution/types.ts +21 -2
- package/src/execution/utils.ts +28 -1
- package/src/helpers/cache.ts +33 -0
- package/src/helpers/index.ts +1 -0
- package/src/helpers/spectate.ts +10 -0
- package/src/index.ts +1 -0
- package/src/model.ts +119 -16
- package/src/plugins/anthropic.ts +159 -20
- package/src/plugins/openai.ts +26 -1
- package/src/plugins/types.ts +42 -5
- package/src/prompt/index.ts +5 -0
- package/src/prompt/plugins.ts +69 -0
- package/src/prompt/render.ts +48 -0
- package/src/prompt/service.ts +194 -0
- package/src/prompt/types.ts +114 -0
- package/src/service.ts +3 -0
- package/src/types.ts +54 -3
- package/src/utils/prompt.ts +33 -0
- package/tests/execution.spec.ts +42 -0
- package/tests/plugins.spec.ts +279 -14
- package/tests/prompt.spec.ts +194 -0
package/src/consts.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { ExecutionEffort } from '@owlmeans/llm-common'
|
|
2
|
-
import type { ModelConfigPatch } from '@owlmeans/llm-common'
|
|
2
|
+
import type { CacheTtl, ModelConfigPatch } from '@owlmeans/llm-common'
|
|
3
3
|
|
|
4
4
|
/** Context-service alias for the {@link LlmService} (model factory / registry). */
|
|
5
5
|
export const LLM_SERVICE = 'owlmeans-llm-service'
|
|
@@ -7,6 +7,9 @@ export const LLM_SERVICE = 'owlmeans-llm-service'
|
|
|
7
7
|
/** Context-service alias for the {@link ExecutionService}. */
|
|
8
8
|
export const EXECUTION_SERVICE = 'owlmeans-llm-execution-service'
|
|
9
9
|
|
|
10
|
+
/** Context-service alias for the {@link PromptService} (skill registry + composition). */
|
|
11
|
+
export const PROMPT_SERVICE = 'owlmeans-llm-prompt-service'
|
|
12
|
+
|
|
10
13
|
/** Default number of attempts a single model call makes before giving up. */
|
|
11
14
|
export const DEFAULT_MODEL_RETRIES = 8
|
|
12
15
|
|
|
@@ -17,9 +20,15 @@ export const DEFAULT_MODEL_RETRIES = 8
|
|
|
17
20
|
* against a provider that accepts the request and then never streams anything (observed
|
|
18
21
|
* with throughput-sorted OpenRouter routing), which would otherwise block forever —
|
|
19
22
|
* `maxRetries` never helps there because the request never errors, it just hangs.
|
|
20
|
-
* Overridable per model via `ModelConfig.streamTimeout
|
|
23
|
+
* Overridable per model via `ModelConfig.streamTimeout`, or for every model at once via
|
|
24
|
+
* `LlmServiceOptions.streamTimeout` where the application composes its context.
|
|
25
|
+
*
|
|
26
|
+
* Three minutes, not longer: the timer resets on every token, so a legitimately long
|
|
27
|
+
* generation is never at risk — this only bounds SILENCE. The cost of a high value is
|
|
28
|
+
* paid entirely by hung calls, and a hang that takes five minutes to notice is a hang
|
|
29
|
+
* that can stall an agent run for half an hour once retries multiply it.
|
|
21
30
|
*/
|
|
22
|
-
export const MODEL_STREAM_TIMEOUT_MS =
|
|
31
|
+
export const MODEL_STREAM_TIMEOUT_MS = 3 * 60 * 1000
|
|
23
32
|
|
|
24
33
|
/**
|
|
25
34
|
* Number of failed attempts after which the retry escalator switches from a role's
|
|
@@ -37,9 +46,38 @@ export const FALLBACK_AFTER_ATTEMPTS = 3
|
|
|
37
46
|
*/
|
|
38
47
|
export const DEFAULT_MAX_OUTPUT_CAP = 3 * 64000
|
|
39
48
|
|
|
40
|
-
/** Provider hard limit on prompt-cache breakpoints (Anthropic). */
|
|
49
|
+
/** Provider hard limit on prompt-cache breakpoints per REQUEST (Anthropic). */
|
|
41
50
|
export const MAX_CACHE_BREAKPOINTS = 4
|
|
42
51
|
|
|
52
|
+
/**
|
|
53
|
+
* Share of {@link MAX_CACHE_BREAKPOINTS} the composed system prompt may spend.
|
|
54
|
+
*
|
|
55
|
+
* Two is all it can use: the only STABLE boundaries are the end of role+skills and the end
|
|
56
|
+
* of the packages block. The trailing context block is volatile and deliberately never
|
|
57
|
+
* marked, so the remaining two breakpoints always stay available to the message prefix.
|
|
58
|
+
*/
|
|
59
|
+
export const MAX_SYSTEM_BREAKPOINTS = 2
|
|
60
|
+
|
|
61
|
+
/** Cache lifetime used when neither the call nor the service asks for another. */
|
|
62
|
+
export const DEFAULT_CACHE_TTL: CacheTtl = '5m'
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Smallest prefix worth marking as cacheable, in tokens. Providers silently refuse to
|
|
66
|
+
* create an entry below their own threshold (1024 tokens on most Claude models, 512 on
|
|
67
|
+
* the newest, 4096 on a few older ones — it is NOT monotonic across generations), so a
|
|
68
|
+
* marker on a short prefix costs nothing but wastes a breakpoint and produces a
|
|
69
|
+
* "caching enabled" log for an entry that was never written. Override per model with
|
|
70
|
+
* `ModelConfig.cacheMinTokens`.
|
|
71
|
+
*/
|
|
72
|
+
export const MIN_CACHEABLE_TOKENS = 1024
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Characters per token used to size a prefix against {@link MIN_CACHEABLE_TOKENS}. A rough
|
|
76
|
+
* average for English prose and TypeScript; only ever used to decide whether marking is
|
|
77
|
+
* worth a breakpoint, never for billing or budgeting.
|
|
78
|
+
*/
|
|
79
|
+
export const CHARS_PER_TOKEN = 4
|
|
80
|
+
|
|
43
81
|
/**
|
|
44
82
|
* Appended to the prompt of `invoke`/`request` when no message already mentions JSON.
|
|
45
83
|
* Some providers refuse or ignore JSON modes unless the word appears in the prompt;
|
|
@@ -85,5 +123,5 @@ export const EFFORT_TABLE: Record<ExecutionEffort, ModelConfigPatch> = {
|
|
|
85
123
|
* without excluding it every `derive`/`escalate`/`withPurpose` would nest another copy.
|
|
86
124
|
*/
|
|
87
125
|
export const COLLABORATOR_KEYS: string[] = [
|
|
88
|
-
'state', 'models', 'model', 'temperatureFactory', 'outputErrors',
|
|
126
|
+
'state', 'models', 'model', 'temperatureFactory', 'outputErrors', 'files', 'prompts',
|
|
89
127
|
]
|
package/src/execution/service.ts
CHANGED
|
@@ -9,7 +9,8 @@ import type {
|
|
|
9
9
|
HelperExecution, TaskExecution, WithExecutionService,
|
|
10
10
|
} from './types.js'
|
|
11
11
|
import {
|
|
12
|
-
composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
|
|
12
|
+
composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
|
|
13
|
+
mergePrompt, resolveRole,
|
|
13
14
|
} from './utils.js'
|
|
14
15
|
|
|
15
16
|
/**
|
|
@@ -50,18 +51,21 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
50
51
|
level: ExecutionLevel.Project,
|
|
51
52
|
purpose: { ...input.purpose },
|
|
52
53
|
policy: { ...input.policy },
|
|
54
|
+
...(input.prompt != null ? { prompt: { ...input.prompt } } : {}),
|
|
53
55
|
}) as S['project'],
|
|
54
56
|
|
|
55
57
|
forTask: (parent, input) => {
|
|
56
|
-
const { effort, phase, data, ...extras } = input
|
|
58
|
+
const { effort, phase, data, prompt, ...extras } = input
|
|
57
59
|
const policy = effort != null
|
|
58
60
|
? mergePolicy(parent.policy, { effort })
|
|
59
61
|
: { ...parent.policy }
|
|
62
|
+
const merged = mergePrompt(parent.prompt, prompt)
|
|
60
63
|
|
|
61
64
|
// Spreading the parent carries every collaborator and domain field forward; the
|
|
62
65
|
// task's own state is composed afterwards, from the seeded resumable fields.
|
|
63
66
|
const taskExec = {
|
|
64
67
|
...parent, ...extras, level: ExecutionLevel.Task, purpose: { ...parent.purpose }, policy,
|
|
68
|
+
...(merged != null ? { prompt: merged } : {}),
|
|
65
69
|
} as unknown as TaskExecution
|
|
66
70
|
;(taskExec as { state: TaskExecutionState }).state = composeTaskState({
|
|
67
71
|
...taskExec,
|
|
@@ -78,9 +82,10 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
78
82
|
},
|
|
79
83
|
|
|
80
84
|
forHelper: (parent, input) => {
|
|
81
|
-
const { role, effort, dedication, ...extras } = input
|
|
85
|
+
const { role, effort, dedication, prompt, ...extras } = input
|
|
82
86
|
const localPolicy = effort != null ? mergePolicy(parent.policy, { effort }) : parent.policy
|
|
83
87
|
const scoped = { ...parent, policy: localPolicy } as S['exec']
|
|
88
|
+
const merged = mergePrompt(parent.prompt, prompt)
|
|
84
89
|
|
|
85
90
|
const helperExec = {
|
|
86
91
|
...parent,
|
|
@@ -90,6 +95,7 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
90
95
|
? { ...parent.purpose, dedication }
|
|
91
96
|
: { ...parent.purpose },
|
|
92
97
|
policy: localPolicy,
|
|
98
|
+
...(merged != null ? { prompt: merged } : {}),
|
|
93
99
|
role: resolveRole(localPolicy, role),
|
|
94
100
|
model: self().model(scoped, role),
|
|
95
101
|
temperatureFactory: self().temperatureFactory(scoped, role),
|
package/src/execution/types.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
2
|
import type { InitializedService } from '@owlmeans/context'
|
|
3
3
|
import type {
|
|
4
|
-
ExecutionEffort, ExecutionLevel, ExecutionState,
|
|
5
|
-
ModelPolicy, ModelRole, TaskExecutionState,
|
|
4
|
+
ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef, LlmPurpose,
|
|
5
|
+
ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
|
|
6
6
|
} from '@owlmeans/llm-common'
|
|
7
7
|
import type { LlmService, TemperatureFactory } from '../types.js'
|
|
8
|
+
import type { PromptService } from '../prompt/types.js'
|
|
8
9
|
|
|
9
10
|
/**
|
|
10
11
|
* Runtime execution = serializable {@link ExecutionState} + attached collaborators.
|
|
@@ -18,6 +19,10 @@ import type { LlmService, TemperatureFactory } from '../types.js'
|
|
|
18
19
|
export interface Execution extends ExecutionState {
|
|
19
20
|
/** Resolver for the model factory — a function so the service can be swapped/cloned. */
|
|
20
21
|
models: () => LlmService
|
|
22
|
+
/** Resolver for the skill registry / prompt composer. Same late-binding rationale. */
|
|
23
|
+
prompts?: () => PromptService
|
|
24
|
+
/** File access offered to prompt plugins. Declared a collaborator, never snapshotted. */
|
|
25
|
+
files?: FileProviderRef
|
|
21
26
|
outputErrors?: boolean
|
|
22
27
|
captureNull?: boolean
|
|
23
28
|
}
|
|
@@ -42,8 +47,12 @@ export interface HelperExecution extends Execution {
|
|
|
42
47
|
|
|
43
48
|
export interface ProjectExecutionInput {
|
|
44
49
|
models: () => LlmService
|
|
50
|
+
prompts?: () => PromptService
|
|
51
|
+
files?: FileProviderRef
|
|
45
52
|
policy: ModelPolicy
|
|
46
53
|
purpose: LlmPurpose
|
|
54
|
+
/** Baseline role and skills for the whole run. */
|
|
55
|
+
prompt?: PromptPolicy
|
|
47
56
|
outputErrors?: boolean
|
|
48
57
|
captureNull?: boolean
|
|
49
58
|
}
|
|
@@ -51,15 +60,23 @@ export interface ProjectExecutionInput {
|
|
|
51
60
|
export interface TaskExecutionInput {
|
|
52
61
|
/** Raise (or lower) the effort tier for this task and everything derived from it. */
|
|
53
62
|
effort?: ExecutionEffort
|
|
63
|
+
/** Skills (and optionally a role) layered on top of the project's. Skills accumulate. */
|
|
64
|
+
prompt?: PromptPolicy
|
|
54
65
|
/** Optional seeds for the resumable task state. */
|
|
55
66
|
phase?: string
|
|
56
67
|
data?: Record<string, unknown>
|
|
57
68
|
}
|
|
58
69
|
|
|
59
70
|
export interface HelperExecutionInput {
|
|
71
|
+
/**
|
|
72
|
+
* Which MODEL to use. Distinct from `prompt.role`, which is the system-prompt text
|
|
73
|
+
* defining the persona — one selects hardware, the other writes the job description.
|
|
74
|
+
*/
|
|
60
75
|
role: ModelRole
|
|
61
76
|
/** Local effort bump without escalating the whole branch. */
|
|
62
77
|
effort?: ExecutionEffort
|
|
78
|
+
/** The helper's persona and its own skills, layered on top of the task's. */
|
|
79
|
+
prompt?: PromptPolicy
|
|
63
80
|
/** Refines `purpose.dedication`. */
|
|
64
81
|
dedication?: string
|
|
65
82
|
}
|
|
@@ -67,6 +84,8 @@ export interface HelperExecutionInput {
|
|
|
67
84
|
/** Collaborators re-attached to a state that was restored from storage. */
|
|
68
85
|
export interface RestoreCollaborators {
|
|
69
86
|
models?: () => LlmService
|
|
87
|
+
prompts?: () => PromptService
|
|
88
|
+
files?: FileProviderRef
|
|
70
89
|
}
|
|
71
90
|
|
|
72
91
|
/**
|
package/src/execution/utils.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { ExecutionLevel } from '@owlmeans/llm-common'
|
|
2
2
|
import type {
|
|
3
3
|
ExecutionEffort, ExecutionState, ModelConfigOverride, ModelConfigPatch,
|
|
4
|
-
ModelPolicy, ModelRole, TaskExecutionState,
|
|
4
|
+
ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
|
|
5
5
|
} from '@owlmeans/llm-common'
|
|
6
6
|
import { EFFORT_TABLE } from '../consts.js'
|
|
7
7
|
import type { Execution, TaskExecution } from './types.js'
|
|
@@ -19,6 +19,33 @@ export const mergePolicy = (base: ModelPolicy, patch: Partial<ModelPolicy>): Mod
|
|
|
19
19
|
: undefined,
|
|
20
20
|
})
|
|
21
21
|
|
|
22
|
+
/**
|
|
23
|
+
* Overlay a prompt policy onto the one inherited from the parent level.
|
|
24
|
+
*
|
|
25
|
+
* Skills ACCUMULATE — a task adds to what the project declared, a helper adds to the
|
|
26
|
+
* task — because that is how a capability set is built up as work narrows. The role is
|
|
27
|
+
* replaced instead: the deepest level that names one owns the persona.
|
|
28
|
+
*
|
|
29
|
+
* The union preserves first-seen order and de-duplicates, so the composed prompt is
|
|
30
|
+
* byte-identical no matter how many levels contributed the same skill.
|
|
31
|
+
*/
|
|
32
|
+
export const mergePrompt = (
|
|
33
|
+
base: PromptPolicy | undefined,
|
|
34
|
+
patch: PromptPolicy | undefined,
|
|
35
|
+
): PromptPolicy | undefined => {
|
|
36
|
+
if (base == null && patch == null) {
|
|
37
|
+
return undefined
|
|
38
|
+
}
|
|
39
|
+
const skills = [...new Set([...(base?.skills ?? []), ...(patch?.skills ?? [])])]
|
|
40
|
+
|
|
41
|
+
return {
|
|
42
|
+
...base,
|
|
43
|
+
...patch,
|
|
44
|
+
...(base?.role != null || patch?.role != null ? { role: patch?.role ?? base?.role } : {}),
|
|
45
|
+
...(skills.length > 0 ? { skills } : {}),
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
22
49
|
/** Apply the policy's role→role remap. */
|
|
23
50
|
export const resolveRole = (policy: ModelPolicy, role: ModelRole): ModelRole =>
|
|
24
51
|
(policy.roleOverrides?.[role] as ModelRole | undefined) ?? role
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import type { UsageMetadata } from '@langchain/core/messages'
|
|
2
|
+
import type { CacheUsage } from '@owlmeans/llm-common'
|
|
3
|
+
|
|
4
|
+
/** LangChain normalizes every provider's cache accounting into these two fields. */
|
|
5
|
+
interface InputTokenDetails {
|
|
6
|
+
cache_read?: number
|
|
7
|
+
cache_creation?: number
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Prompt-cache accounting for one completion.
|
|
12
|
+
*
|
|
13
|
+
* This is the only honest answer to "is caching actually working". A composed prefix can
|
|
14
|
+
* look perfectly stable and still miss on every call — a stray timestamp, a set iterated
|
|
15
|
+
* in a different order, a tool list rebuilt per request. If `read` stays at zero across
|
|
16
|
+
* repeated calls that share a prefix, something is invalidating it; diff the rendered
|
|
17
|
+
* blocks between two calls to find out what.
|
|
18
|
+
*/
|
|
19
|
+
export const readCacheUsage = (message: { usage_metadata?: UsageMetadata }): CacheUsage => {
|
|
20
|
+
const usage = message.usage_metadata
|
|
21
|
+
const details = usage?.input_token_details as InputTokenDetails | undefined
|
|
22
|
+
|
|
23
|
+
return {
|
|
24
|
+
read: details?.cache_read ?? 0,
|
|
25
|
+
creation: details?.cache_creation ?? 0,
|
|
26
|
+
input: usage?.input_tokens ?? 0,
|
|
27
|
+
output: usage?.output_tokens ?? 0,
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** `true` when the provider reported any cache activity at all. */
|
|
32
|
+
export const hasCacheActivity = (usage: CacheUsage): boolean =>
|
|
33
|
+
usage.read > 0 || usage.creation > 0
|
package/src/helpers/index.ts
CHANGED
package/src/helpers/spectate.ts
CHANGED
|
@@ -3,6 +3,7 @@ import type { UsageMetadata } from '@langchain/core/messages'
|
|
|
3
3
|
import { SpectatorContentType } from '@owlmeans/llm-common'
|
|
4
4
|
import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
|
|
5
5
|
import type { LlmSpectator, ModelInputItem } from '../types.js'
|
|
6
|
+
import { hasCacheActivity, readCacheUsage } from './cache.js'
|
|
6
7
|
|
|
7
8
|
/** Normalize one prompt message into the spectator's storage shape. */
|
|
8
9
|
const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
|
|
@@ -63,5 +64,14 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
|
|
|
63
64
|
completion.content = message.content as unknown as string
|
|
64
65
|
}
|
|
65
66
|
|
|
67
|
+
// Silent unless the provider reported cache activity, so it costs nothing when
|
|
68
|
+
// caching is off — and is the one signal that tells a stable prefix from a broken one.
|
|
69
|
+
const cache = readCacheUsage(message)
|
|
70
|
+
if (hasCacheActivity(cache)) {
|
|
71
|
+
console.log(
|
|
72
|
+
`Prompt cache [${action}]: read ${cache.read}, written ${cache.creation}, uncached ${cache.input}`
|
|
73
|
+
)
|
|
74
|
+
}
|
|
75
|
+
|
|
66
76
|
return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
|
|
67
77
|
}
|
package/src/index.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './model.js'
|
|
|
6
6
|
export * from './service.js'
|
|
7
7
|
export * from './helpers/index.js'
|
|
8
8
|
export * from './execution/index.js'
|
|
9
|
+
export * from './prompt/index.js'
|
|
9
10
|
export type * from './plugins/types.js'
|
|
10
11
|
export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
|
|
11
12
|
export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
|
package/src/model.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { Ajv } from 'ajv'
|
|
2
2
|
import type { JSONSchemaType } from 'ajv'
|
|
3
|
-
import { AIMessage } from '@langchain/core/messages'
|
|
4
|
-
import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
|
|
3
|
+
import { AIMessage, BaseMessage } from '@langchain/core/messages'
|
|
4
|
+
import type { AIMessageChunk, MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
|
|
5
5
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
6
6
|
import { StructuredMode } from '@owlmeans/llm-common'
|
|
7
7
|
import type { NullKind } from '@owlmeans/llm-common'
|
|
8
8
|
import {
|
|
9
|
-
DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
|
|
9
|
+
DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
|
|
10
10
|
} from './consts.js'
|
|
11
11
|
import { LlmModelError } from './errors.js'
|
|
12
12
|
import { pluginFor, pluginOf } from './plugins/index.js'
|
|
@@ -18,7 +18,7 @@ import { spectate } from './helpers/spectate.js'
|
|
|
18
18
|
import { idleTimeout, readConfig } from './utils/config.js'
|
|
19
19
|
import { reportNull } from './utils/null-report.js'
|
|
20
20
|
import type { NullReportParams } from './utils/null-report.js'
|
|
21
|
-
import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
|
|
21
|
+
import { applyNoThink, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
|
|
22
22
|
import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
|
|
23
23
|
import { streamWithDeadline } from './utils/stream.js'
|
|
24
24
|
import type {
|
|
@@ -28,6 +28,48 @@ import type {
|
|
|
28
28
|
|
|
29
29
|
type StreamOptions = Parameters<BaseChatModel['stream']>[1]
|
|
30
30
|
|
|
31
|
+
const isSystem = (msg: MessageFieldWithRole): boolean =>
|
|
32
|
+
msg instanceof BaseMessage ? msg.getType() === 'system' : `${msg.role}` === 'system'
|
|
33
|
+
|
|
34
|
+
const textOf = (content: MessageContent | undefined): string => {
|
|
35
|
+
if (typeof content === 'string') {
|
|
36
|
+
return content.trim()
|
|
37
|
+
}
|
|
38
|
+
if (Array.isArray(content)) {
|
|
39
|
+
return content
|
|
40
|
+
.map(part => {
|
|
41
|
+
const text = (part as unknown as { text?: unknown }).text
|
|
42
|
+
return typeof text === 'string' ? text : ''
|
|
43
|
+
})
|
|
44
|
+
.filter(text => text !== '')
|
|
45
|
+
.join('\n\n')
|
|
46
|
+
.trim()
|
|
47
|
+
}
|
|
48
|
+
return ''
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Detach the caller's LEADING system messages and return their text.
|
|
53
|
+
*
|
|
54
|
+
* They are re-emitted as the `Context` block of the composed prompt, which is what keeps
|
|
55
|
+
* a caller that still builds its own `SystemMessage` working unchanged — the text simply
|
|
56
|
+
* travels a different route and lands in the same place. Only the leading run is taken:
|
|
57
|
+
* a system message deliberately placed mid-conversation is an operator instruction whose
|
|
58
|
+
* position carries meaning, and moving it would change what the model sees.
|
|
59
|
+
*/
|
|
60
|
+
const takeLeadingSystem = (msgs: MessageFieldWithRole[]): string[] => {
|
|
61
|
+
const carried: string[] = []
|
|
62
|
+
while (msgs.length > 0 && isSystem(msgs[0])) {
|
|
63
|
+
const [msg] = msgs.splice(0, 1)
|
|
64
|
+
const text = textOf(msg.content)
|
|
65
|
+
if (text !== '') {
|
|
66
|
+
carried.push(text)
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
return carried
|
|
71
|
+
}
|
|
72
|
+
|
|
31
73
|
/**
|
|
32
74
|
* Build the four-method model API on top of a LangChain chat model.
|
|
33
75
|
*
|
|
@@ -43,6 +85,9 @@ export const makeLlmModel = ({
|
|
|
43
85
|
captureNull = false,
|
|
44
86
|
retries = DEFAULT_MODEL_RETRIES,
|
|
45
87
|
purpose,
|
|
88
|
+
prompt,
|
|
89
|
+
prompts,
|
|
90
|
+
files,
|
|
46
91
|
}: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
|
|
47
92
|
|
|
48
93
|
const ajv = new Ajv({ strict: false })
|
|
@@ -53,15 +98,67 @@ export const makeLlmModel = ({
|
|
|
53
98
|
const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
|
|
54
99
|
const timeout = idleTimeout(config)
|
|
55
100
|
|
|
56
|
-
/**
|
|
57
|
-
|
|
101
|
+
/**
|
|
102
|
+
* Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
|
|
103
|
+
* dependency order.
|
|
104
|
+
*
|
|
105
|
+
* With no prompt service wired this is exactly what it always was — the caller's
|
|
106
|
+
* messages, a JSON nudge, `/no_think`, cache markers. With one, the caller's leading
|
|
107
|
+
* system text is folded into a composed prompt whose stable sections come first, which
|
|
108
|
+
* is the whole point: a prompt cache is a PREFIX match, so the bytes every call shares
|
|
109
|
+
* have to be physically ahead of the bytes that differ.
|
|
110
|
+
*/
|
|
111
|
+
const prepare = async (
|
|
112
|
+
input: ModelInput,
|
|
113
|
+
action: string,
|
|
114
|
+
useCache: boolean,
|
|
115
|
+
cacheMax: number,
|
|
116
|
+
json: boolean,
|
|
117
|
+
callSkills?: string[],
|
|
118
|
+
): Promise<MessageFieldWithRole[]> => {
|
|
58
119
|
const msgs = normalizeInput(input)
|
|
120
|
+
// The caller may hand back messages this pipeline marked on a PREVIOUS call — the
|
|
121
|
+
// markers live on its own objects. The budget is per request, so clear them and
|
|
122
|
+
// re-place our own below; otherwise they accumulate until the provider 400s.
|
|
123
|
+
stripCacheMarkers(msgs)
|
|
124
|
+
let reserved = 0
|
|
125
|
+
|
|
126
|
+
if (prompts != null) {
|
|
127
|
+
const carried = takeLeadingSystem(msgs)
|
|
128
|
+
const composed = await prompts().compose(
|
|
129
|
+
{
|
|
130
|
+
...prompt,
|
|
131
|
+
context: [...(prompt?.context ?? []), ...carried],
|
|
132
|
+
callSkills: callSkills ?? prompt?.callSkills,
|
|
133
|
+
},
|
|
134
|
+
msgs,
|
|
135
|
+
{ model, provider: plugin, purpose, action, cacheMax, files },
|
|
136
|
+
)
|
|
137
|
+
if (composed.system != null) {
|
|
138
|
+
msgs.unshift(composed.system)
|
|
139
|
+
reserved = composed.breakpoints
|
|
140
|
+
} else {
|
|
141
|
+
// Defensive: nothing was contributed, so hand the caller's own text straight back.
|
|
142
|
+
for (let i = carried.length - 1; i >= 0; i--) {
|
|
143
|
+
msgs.unshift({ role: 'system', content: carried[i] })
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
59
148
|
if (json) ensureJsonMention(msgs)
|
|
60
149
|
applyNoThink(msgs, config.disableThinking)
|
|
61
150
|
// Cache markers replace string content with content blocks, so they must go last.
|
|
62
|
-
|
|
63
|
-
|
|
151
|
+
const ttl = prompt?.cacheTtl
|
|
152
|
+
const marked = plugin?.patchCache?.(msgs, {
|
|
153
|
+
model, useCache, cacheMax, reserved, ...(ttl != null ? { ttl } : {}),
|
|
154
|
+
})
|
|
155
|
+
if (reserved > 0 || marked === true) {
|
|
156
|
+
console.log(
|
|
157
|
+
`Prompt caching for ${plugin?.type}: ${reserved} system breakpoint(s)`
|
|
158
|
+
+ `${marked === true ? ', 1 message breakpoint' : ''}`
|
|
159
|
+
)
|
|
64
160
|
}
|
|
161
|
+
|
|
65
162
|
return msgs
|
|
66
163
|
}
|
|
67
164
|
|
|
@@ -197,8 +294,11 @@ export const makeLlmModel = ({
|
|
|
197
294
|
}
|
|
198
295
|
|
|
199
296
|
const helper: LlmModel = {
|
|
200
|
-
ask: async (
|
|
201
|
-
|
|
297
|
+
ask: async (
|
|
298
|
+
input,
|
|
299
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmAskOptions
|
|
300
|
+
) => {
|
|
301
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
202
302
|
return withRetry({ retries, outputErrors }, async i => {
|
|
203
303
|
const refined = refineModel(i)
|
|
204
304
|
console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
|
|
@@ -242,8 +342,11 @@ export const makeLlmModel = ({
|
|
|
242
342
|
})
|
|
243
343
|
},
|
|
244
344
|
|
|
245
|
-
talk: async (
|
|
246
|
-
|
|
345
|
+
talk: async (
|
|
346
|
+
input,
|
|
347
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmTalkOptions
|
|
348
|
+
) => {
|
|
349
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
247
350
|
return withRetry({ retries, outputErrors }, async i => {
|
|
248
351
|
const refined = refineModel(i)
|
|
249
352
|
console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
|
|
@@ -277,9 +380,9 @@ export const makeLlmModel = ({
|
|
|
277
380
|
invoke: async <T>(
|
|
278
381
|
input: ModelInput,
|
|
279
382
|
schema: JSONSchemaType<T>,
|
|
280
|
-
{ temperature, ref, filter, action, useCache = false, cacheMax =
|
|
383
|
+
{ temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmInvokeOptions<T>
|
|
281
384
|
) => {
|
|
282
|
-
const msgs = prepare(input, useCache, cacheMax, true)
|
|
385
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
283
386
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
284
387
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
285
388
|
|
|
@@ -325,9 +428,9 @@ export const makeLlmModel = ({
|
|
|
325
428
|
request: async <T>(
|
|
326
429
|
input: ModelInput,
|
|
327
430
|
schema: JSONSchemaType<T>,
|
|
328
|
-
{ ref, filter, action, useCache = false, cacheMax =
|
|
431
|
+
{ ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmRequestOptions
|
|
329
432
|
) => {
|
|
330
|
-
const msgs = prepare(input, useCache, cacheMax, true)
|
|
433
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
331
434
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
332
435
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
333
436
|
|