@owlmeans/llm 0.1.14 → 0.1.16-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/agent-meta/instructions/llm-prompt-caching.instructions.md +100 -0
  2. package/agent-meta/instructions/llm.instructions.md +13 -3
  3. package/agent-meta/manifest.json +16 -2
  4. package/agent-meta/skills/llm/SKILL.md +54 -4
  5. package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
  6. package/build/consts.d.ts +36 -3
  7. package/build/consts.d.ts.map +1 -1
  8. package/build/consts.js +37 -4
  9. package/build/consts.js.map +1 -1
  10. package/build/execution/service.d.ts.map +1 -1
  11. package/build/execution/service.js +8 -3
  12. package/build/execution/service.js.map +1 -1
  13. package/build/execution/types.d.ts +20 -1
  14. package/build/execution/types.d.ts.map +1 -1
  15. package/build/execution/utils.d.ts +12 -1
  16. package/build/execution/utils.d.ts.map +1 -1
  17. package/build/execution/utils.js +22 -0
  18. package/build/execution/utils.js.map +1 -1
  19. package/build/helpers/cache.d.ts +17 -0
  20. package/build/helpers/cache.d.ts.map +1 -0
  21. package/build/helpers/cache.js +22 -0
  22. package/build/helpers/cache.js.map +1 -0
  23. package/build/helpers/index.d.ts +1 -0
  24. package/build/helpers/index.d.ts.map +1 -1
  25. package/build/helpers/index.js +1 -0
  26. package/build/helpers/index.js.map +1 -1
  27. package/build/helpers/spectate.d.ts.map +1 -1
  28. package/build/helpers/spectate.js +7 -0
  29. package/build/helpers/spectate.js.map +1 -1
  30. package/build/index.d.ts +1 -0
  31. package/build/index.d.ts.map +1 -1
  32. package/build/index.js +1 -0
  33. package/build/index.js.map +1 -1
  34. package/build/model.d.ts +1 -1
  35. package/build/model.d.ts.map +1 -1
  36. package/build/model.js +90 -16
  37. package/build/model.js.map +1 -1
  38. package/build/plugins/anthropic.d.ts.map +1 -1
  39. package/build/plugins/anthropic.js +136 -21
  40. package/build/plugins/anthropic.js.map +1 -1
  41. package/build/plugins/openai.d.ts +9 -0
  42. package/build/plugins/openai.d.ts.map +1 -1
  43. package/build/plugins/openai.js +23 -1
  44. package/build/plugins/openai.js.map +1 -1
  45. package/build/plugins/types.d.ts +38 -5
  46. package/build/plugins/types.d.ts.map +1 -1
  47. package/build/prompt/index.d.ts +5 -0
  48. package/build/prompt/index.d.ts.map +1 -0
  49. package/build/prompt/index.js +4 -0
  50. package/build/prompt/index.js.map +1 -0
  51. package/build/prompt/plugins.d.ts +24 -0
  52. package/build/prompt/plugins.d.ts.map +1 -0
  53. package/build/prompt/plugins.js +64 -0
  54. package/build/prompt/plugins.js.map +1 -0
  55. package/build/prompt/render.d.ts +28 -0
  56. package/build/prompt/render.d.ts.map +1 -0
  57. package/build/prompt/render.js +39 -0
  58. package/build/prompt/render.js.map +1 -0
  59. package/build/prompt/service.d.ts +16 -0
  60. package/build/prompt/service.d.ts.map +1 -0
  61. package/build/prompt/service.js +145 -0
  62. package/build/prompt/service.js.map +1 -0
  63. package/build/prompt/types.d.ts +101 -0
  64. package/build/prompt/types.d.ts.map +1 -0
  65. package/build/prompt/types.js +2 -0
  66. package/build/prompt/types.js.map +1 -0
  67. package/build/service.d.ts.map +1 -1
  68. package/build/service.js +3 -0
  69. package/build/service.js.map +1 -1
  70. package/build/types.d.ts +53 -3
  71. package/build/types.d.ts.map +1 -1
  72. package/build/utils/prompt.d.ts +14 -0
  73. package/build/utils/prompt.d.ts.map +1 -1
  74. package/build/utils/prompt.js +32 -0
  75. package/build/utils/prompt.js.map +1 -1
  76. package/package.json +13 -6
  77. package/src/consts.ts +43 -5
  78. package/src/execution/service.ts +9 -3
  79. package/src/execution/types.ts +21 -2
  80. package/src/execution/utils.ts +28 -1
  81. package/src/helpers/cache.ts +33 -0
  82. package/src/helpers/index.ts +1 -0
  83. package/src/helpers/spectate.ts +10 -0
  84. package/src/index.ts +1 -0
  85. package/src/model.ts +119 -16
  86. package/src/plugins/anthropic.ts +159 -20
  87. package/src/plugins/openai.ts +26 -1
  88. package/src/plugins/types.ts +42 -5
  89. package/src/prompt/index.ts +5 -0
  90. package/src/prompt/plugins.ts +69 -0
  91. package/src/prompt/render.ts +48 -0
  92. package/src/prompt/service.ts +194 -0
  93. package/src/prompt/types.ts +114 -0
  94. package/src/service.ts +3 -0
  95. package/src/types.ts +54 -3
  96. package/src/utils/prompt.ts +33 -0
  97. package/tests/execution.spec.ts +42 -0
  98. package/tests/plugins.spec.ts +279 -14
  99. package/tests/prompt.spec.ts +194 -0
package/src/consts.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { ExecutionEffort } from '@owlmeans/llm-common'
2
- import type { ModelConfigPatch } from '@owlmeans/llm-common'
2
+ import type { CacheTtl, ModelConfigPatch } from '@owlmeans/llm-common'
3
3
 
4
4
  /** Context-service alias for the {@link LlmService} (model factory / registry). */
5
5
  export const LLM_SERVICE = 'owlmeans-llm-service'
@@ -7,6 +7,9 @@ export const LLM_SERVICE = 'owlmeans-llm-service'
7
7
  /** Context-service alias for the {@link ExecutionService}. */
8
8
  export const EXECUTION_SERVICE = 'owlmeans-llm-execution-service'
9
9
 
10
+ /** Context-service alias for the {@link PromptService} (skill registry + composition). */
11
+ export const PROMPT_SERVICE = 'owlmeans-llm-prompt-service'
12
+
10
13
  /** Default number of attempts a single model call makes before giving up. */
11
14
  export const DEFAULT_MODEL_RETRIES = 8
12
15
 
@@ -17,9 +20,15 @@ export const DEFAULT_MODEL_RETRIES = 8
17
20
  * against a provider that accepts the request and then never streams anything (observed
18
21
  * with throughput-sorted OpenRouter routing), which would otherwise block forever —
19
22
  * `maxRetries` never helps there because the request never errors, it just hangs.
20
- * Overridable per model via `ModelConfig.streamTimeout`.
23
+ * Overridable per model via `ModelConfig.streamTimeout`, or for every model at once via
24
+ * `LlmServiceOptions.streamTimeout` where the application composes its context.
25
+ *
26
+ * Three minutes, not longer: the timer resets on every token, so a legitimately long
27
+ * generation is never at risk — this only bounds SILENCE. The cost of a high value is
28
+ * paid entirely by hung calls, and a hang that takes five minutes to notice is a hang
29
+ * that can stall an agent run for half an hour once retries multiply it.
21
30
  */
22
- export const MODEL_STREAM_TIMEOUT_MS = 5 * 60 * 1000
31
+ export const MODEL_STREAM_TIMEOUT_MS = 3 * 60 * 1000
23
32
 
24
33
  /**
25
34
  * Number of failed attempts after which the retry escalator switches from a role's
@@ -37,9 +46,38 @@ export const FALLBACK_AFTER_ATTEMPTS = 3
37
46
  */
38
47
  export const DEFAULT_MAX_OUTPUT_CAP = 3 * 64000
39
48
 
40
- /** Provider hard limit on prompt-cache breakpoints (Anthropic). */
49
+ /** Provider hard limit on prompt-cache breakpoints per REQUEST (Anthropic). */
41
50
  export const MAX_CACHE_BREAKPOINTS = 4
42
51
 
52
+ /**
53
+ * Share of {@link MAX_CACHE_BREAKPOINTS} the composed system prompt may spend.
54
+ *
55
+ * Two is all it can use: the only STABLE boundaries are the end of role+skills and the end
56
+ * of the packages block. The trailing context block is volatile and deliberately never
57
+ * marked, so the remaining two breakpoints always stay available to the message prefix.
58
+ */
59
+ export const MAX_SYSTEM_BREAKPOINTS = 2
60
+
61
+ /** Cache lifetime used when neither the call nor the service asks for another. */
62
+ export const DEFAULT_CACHE_TTL: CacheTtl = '5m'
63
+
64
+ /**
65
+ * Smallest prefix worth marking as cacheable, in tokens. Providers silently refuse to
66
+ * create an entry below their own threshold (1024 tokens on most Claude models, 512 on
67
+ * the newest, 4096 on a few older ones — it is NOT monotonic across generations), so a
68
+ * marker on a short prefix costs nothing but wastes a breakpoint and produces a
69
+ * "caching enabled" log for an entry that was never written. Override per model with
70
+ * `ModelConfig.cacheMinTokens`.
71
+ */
72
+ export const MIN_CACHEABLE_TOKENS = 1024
73
+
74
+ /**
75
+ * Characters per token used to size a prefix against {@link MIN_CACHEABLE_TOKENS}. A rough
76
+ * average for English prose and TypeScript; only ever used to decide whether marking is
77
+ * worth a breakpoint, never for billing or budgeting.
78
+ */
79
+ export const CHARS_PER_TOKEN = 4
80
+
43
81
  /**
44
82
  * Appended to the prompt of `invoke`/`request` when no message already mentions JSON.
45
83
  * Some providers refuse or ignore JSON modes unless the word appears in the prompt;
@@ -85,5 +123,5 @@ export const EFFORT_TABLE: Record<ExecutionEffort, ModelConfigPatch> = {
85
123
  * without excluding it every `derive`/`escalate`/`withPurpose` would nest another copy.
86
124
  */
87
125
  export const COLLABORATOR_KEYS: string[] = [
88
- 'state', 'models', 'model', 'temperatureFactory', 'outputErrors',
126
+ 'state', 'models', 'model', 'temperatureFactory', 'outputErrors', 'files', 'prompts',
89
127
  ]
@@ -9,7 +9,8 @@ import type {
9
9
  HelperExecution, TaskExecution, WithExecutionService,
10
10
  } from './types.js'
11
11
  import {
12
- composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy, resolveRole,
12
+ composeExecState, composeTaskState, effortPatch, freeze, mergeOverride, mergePolicy,
13
+ mergePrompt, resolveRole,
13
14
  } from './utils.js'
14
15
 
15
16
  /**
@@ -50,18 +51,21 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
50
51
  level: ExecutionLevel.Project,
51
52
  purpose: { ...input.purpose },
52
53
  policy: { ...input.policy },
54
+ ...(input.prompt != null ? { prompt: { ...input.prompt } } : {}),
53
55
  }) as S['project'],
54
56
 
55
57
  forTask: (parent, input) => {
56
- const { effort, phase, data, ...extras } = input
58
+ const { effort, phase, data, prompt, ...extras } = input
57
59
  const policy = effort != null
58
60
  ? mergePolicy(parent.policy, { effort })
59
61
  : { ...parent.policy }
62
+ const merged = mergePrompt(parent.prompt, prompt)
60
63
 
61
64
  // Spreading the parent carries every collaborator and domain field forward; the
62
65
  // task's own state is composed afterwards, from the seeded resumable fields.
63
66
  const taskExec = {
64
67
  ...parent, ...extras, level: ExecutionLevel.Task, purpose: { ...parent.purpose }, policy,
68
+ ...(merged != null ? { prompt: merged } : {}),
65
69
  } as unknown as TaskExecution
66
70
  ;(taskExec as { state: TaskExecutionState }).state = composeTaskState({
67
71
  ...taskExec,
@@ -78,9 +82,10 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
78
82
  },
79
83
 
80
84
  forHelper: (parent, input) => {
81
- const { role, effort, dedication, ...extras } = input
85
+ const { role, effort, dedication, prompt, ...extras } = input
82
86
  const localPolicy = effort != null ? mergePolicy(parent.policy, { effort }) : parent.policy
83
87
  const scoped = { ...parent, policy: localPolicy } as S['exec']
88
+ const merged = mergePrompt(parent.prompt, prompt)
84
89
 
85
90
  const helperExec = {
86
91
  ...parent,
@@ -90,6 +95,7 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
90
95
  ? { ...parent.purpose, dedication }
91
96
  : { ...parent.purpose },
92
97
  policy: localPolicy,
98
+ ...(merged != null ? { prompt: merged } : {}),
93
99
  role: resolveRole(localPolicy, role),
94
100
  model: self().model(scoped, role),
95
101
  temperatureFactory: self().temperatureFactory(scoped, role),
@@ -1,10 +1,11 @@
1
1
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
2
  import type { InitializedService } from '@owlmeans/context'
3
3
  import type {
4
- ExecutionEffort, ExecutionLevel, ExecutionState, LlmPurpose, ModelConfigOverride,
5
- ModelPolicy, ModelRole, TaskExecutionState,
4
+ ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef, LlmPurpose,
5
+ ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
6
6
  } from '@owlmeans/llm-common'
7
7
  import type { LlmService, TemperatureFactory } from '../types.js'
8
+ import type { PromptService } from '../prompt/types.js'
8
9
 
9
10
  /**
10
11
  * Runtime execution = serializable {@link ExecutionState} + attached collaborators.
@@ -18,6 +19,10 @@ import type { LlmService, TemperatureFactory } from '../types.js'
18
19
  export interface Execution extends ExecutionState {
19
20
  /** Resolver for the model factory — a function so the service can be swapped/cloned. */
20
21
  models: () => LlmService
22
+ /** Resolver for the skill registry / prompt composer. Same late-binding rationale. */
23
+ prompts?: () => PromptService
24
+ /** File access offered to prompt plugins. Declared a collaborator, never snapshotted. */
25
+ files?: FileProviderRef
21
26
  outputErrors?: boolean
22
27
  captureNull?: boolean
23
28
  }
@@ -42,8 +47,12 @@ export interface HelperExecution extends Execution {
42
47
 
43
48
  export interface ProjectExecutionInput {
44
49
  models: () => LlmService
50
+ prompts?: () => PromptService
51
+ files?: FileProviderRef
45
52
  policy: ModelPolicy
46
53
  purpose: LlmPurpose
54
+ /** Baseline role and skills for the whole run. */
55
+ prompt?: PromptPolicy
47
56
  outputErrors?: boolean
48
57
  captureNull?: boolean
49
58
  }
@@ -51,15 +60,23 @@ export interface ProjectExecutionInput {
51
60
  export interface TaskExecutionInput {
52
61
  /** Raise (or lower) the effort tier for this task and everything derived from it. */
53
62
  effort?: ExecutionEffort
63
+ /** Skills (and optionally a role) layered on top of the project's. Skills accumulate. */
64
+ prompt?: PromptPolicy
54
65
  /** Optional seeds for the resumable task state. */
55
66
  phase?: string
56
67
  data?: Record<string, unknown>
57
68
  }
58
69
 
59
70
  export interface HelperExecutionInput {
71
+ /**
72
+ * Which MODEL to use. Distinct from `prompt.role`, which is the system-prompt text
73
+ * defining the persona — one selects hardware, the other writes the job description.
74
+ */
60
75
  role: ModelRole
61
76
  /** Local effort bump without escalating the whole branch. */
62
77
  effort?: ExecutionEffort
78
+ /** The helper's persona and its own skills, layered on top of the task's. */
79
+ prompt?: PromptPolicy
63
80
  /** Refines `purpose.dedication`. */
64
81
  dedication?: string
65
82
  }
@@ -67,6 +84,8 @@ export interface HelperExecutionInput {
67
84
  /** Collaborators re-attached to a state that was restored from storage. */
68
85
  export interface RestoreCollaborators {
69
86
  models?: () => LlmService
87
+ prompts?: () => PromptService
88
+ files?: FileProviderRef
70
89
  }
71
90
 
72
91
  /**
@@ -1,7 +1,7 @@
1
1
  import { ExecutionLevel } from '@owlmeans/llm-common'
2
2
  import type {
3
3
  ExecutionEffort, ExecutionState, ModelConfigOverride, ModelConfigPatch,
4
- ModelPolicy, ModelRole, TaskExecutionState,
4
+ ModelPolicy, ModelRole, PromptPolicy, TaskExecutionState,
5
5
  } from '@owlmeans/llm-common'
6
6
  import { EFFORT_TABLE } from '../consts.js'
7
7
  import type { Execution, TaskExecution } from './types.js'
@@ -19,6 +19,33 @@ export const mergePolicy = (base: ModelPolicy, patch: Partial<ModelPolicy>): Mod
19
19
  : undefined,
20
20
  })
21
21
 
22
+ /**
23
+ * Overlay a prompt policy onto the one inherited from the parent level.
24
+ *
25
+ * Skills ACCUMULATE — a task adds to what the project declared, a helper adds to the
26
+ * task — because that is how a capability set is built up as work narrows. The role is
27
+ * replaced instead: the deepest level that names one owns the persona.
28
+ *
29
+ * The union preserves first-seen order and de-duplicates, so the composed prompt is
30
+ * byte-identical no matter how many levels contributed the same skill.
31
+ */
32
+ export const mergePrompt = (
33
+ base: PromptPolicy | undefined,
34
+ patch: PromptPolicy | undefined,
35
+ ): PromptPolicy | undefined => {
36
+ if (base == null && patch == null) {
37
+ return undefined
38
+ }
39
+ const skills = [...new Set([...(base?.skills ?? []), ...(patch?.skills ?? [])])]
40
+
41
+ return {
42
+ ...base,
43
+ ...patch,
44
+ ...(base?.role != null || patch?.role != null ? { role: patch?.role ?? base?.role } : {}),
45
+ ...(skills.length > 0 ? { skills } : {}),
46
+ }
47
+ }
48
+
22
49
  /** Apply the policy's role→role remap. */
23
50
  export const resolveRole = (policy: ModelPolicy, role: ModelRole): ModelRole =>
24
51
  (policy.roleOverrides?.[role] as ModelRole | undefined) ?? role
@@ -0,0 +1,33 @@
1
+ import type { UsageMetadata } from '@langchain/core/messages'
2
+ import type { CacheUsage } from '@owlmeans/llm-common'
3
+
4
+ /** LangChain normalizes every provider's cache accounting into these two fields. */
5
+ interface InputTokenDetails {
6
+ cache_read?: number
7
+ cache_creation?: number
8
+ }
9
+
10
+ /**
11
+ * Prompt-cache accounting for one completion.
12
+ *
13
+ * This is the only honest answer to "is caching actually working". A composed prefix can
14
+ * look perfectly stable and still miss on every call — a stray timestamp, a set iterated
15
+ * in a different order, a tool list rebuilt per request. If `read` stays at zero across
16
+ * repeated calls that share a prefix, something is invalidating it; diff the rendered
17
+ * blocks between two calls to find out what.
18
+ */
19
+ export const readCacheUsage = (message: { usage_metadata?: UsageMetadata }): CacheUsage => {
20
+ const usage = message.usage_metadata
21
+ const details = usage?.input_token_details as InputTokenDetails | undefined
22
+
23
+ return {
24
+ read: details?.cache_read ?? 0,
25
+ creation: details?.cache_creation ?? 0,
26
+ input: usage?.input_tokens ?? 0,
27
+ output: usage?.output_tokens ?? 0,
28
+ }
29
+ }
30
+
31
+ /** `true` when the provider reported any cache activity at all. */
32
+ export const hasCacheActivity = (usage: CacheUsage): boolean =>
33
+ usage.read > 0 || usage.creation > 0
@@ -3,3 +3,4 @@ export * from './retry.js'
3
3
  export * from './json.js'
4
4
  export * from './messages.js'
5
5
  export * from './spectate.js'
6
+ export * from './cache.js'
@@ -3,6 +3,7 @@ import type { UsageMetadata } from '@langchain/core/messages'
3
3
  import { SpectatorContentType } from '@owlmeans/llm-common'
4
4
  import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
5
5
  import type { LlmSpectator, ModelInputItem } from '../types.js'
6
+ import { hasCacheActivity, readCacheUsage } from './cache.js'
6
7
 
7
8
  /** Normalize one prompt message into the spectator's storage shape. */
8
9
  const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
@@ -63,5 +64,14 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
63
64
  completion.content = message.content as unknown as string
64
65
  }
65
66
 
67
+ // Silent unless the provider reported cache activity, so it costs nothing when
68
+ // caching is off — and is the one signal that tells a stable prefix from a broken one.
69
+ const cache = readCacheUsage(message)
70
+ if (hasCacheActivity(cache)) {
71
+ console.log(
72
+ `Prompt cache [${action}]: read ${cache.read}, written ${cache.creation}, uncached ${cache.input}`
73
+ )
74
+ }
75
+
66
76
  return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
67
77
  }
package/src/index.ts CHANGED
@@ -6,6 +6,7 @@ export * from './model.js'
6
6
  export * from './service.js'
7
7
  export * from './helpers/index.js'
8
8
  export * from './execution/index.js'
9
+ export * from './prompt/index.js'
9
10
  export type * from './plugins/types.js'
10
11
  export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
11
12
  export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
package/src/model.ts CHANGED
@@ -1,12 +1,12 @@
1
1
  import { Ajv } from 'ajv'
2
2
  import type { JSONSchemaType } from 'ajv'
3
- import { AIMessage } from '@langchain/core/messages'
4
- import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
3
+ import { AIMessage, BaseMessage } from '@langchain/core/messages'
4
+ import type { AIMessageChunk, MessageContent, MessageFieldWithRole } from '@langchain/core/messages'
5
5
  import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
6
6
  import { StructuredMode } from '@owlmeans/llm-common'
7
7
  import type { NullKind } from '@owlmeans/llm-common'
8
8
  import {
9
- DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
9
+ DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
10
10
  } from './consts.js'
11
11
  import { LlmModelError } from './errors.js'
12
12
  import { pluginFor, pluginOf } from './plugins/index.js'
@@ -18,7 +18,7 @@ import { spectate } from './helpers/spectate.js'
18
18
  import { idleTimeout, readConfig } from './utils/config.js'
19
19
  import { reportNull } from './utils/null-report.js'
20
20
  import type { NullReportParams } from './utils/null-report.js'
21
- import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
21
+ import { applyNoThink, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
22
22
  import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
23
23
  import { streamWithDeadline } from './utils/stream.js'
24
24
  import type {
@@ -28,6 +28,48 @@ import type {
28
28
 
29
29
  type StreamOptions = Parameters<BaseChatModel['stream']>[1]
30
30
 
31
+ const isSystem = (msg: MessageFieldWithRole): boolean =>
32
+ msg instanceof BaseMessage ? msg.getType() === 'system' : `${msg.role}` === 'system'
33
+
34
+ const textOf = (content: MessageContent | undefined): string => {
35
+ if (typeof content === 'string') {
36
+ return content.trim()
37
+ }
38
+ if (Array.isArray(content)) {
39
+ return content
40
+ .map(part => {
41
+ const text = (part as unknown as { text?: unknown }).text
42
+ return typeof text === 'string' ? text : ''
43
+ })
44
+ .filter(text => text !== '')
45
+ .join('\n\n')
46
+ .trim()
47
+ }
48
+ return ''
49
+ }
50
+
51
+ /**
52
+ * Detach the caller's LEADING system messages and return their text.
53
+ *
54
+ * They are re-emitted as the `Context` block of the composed prompt, which is what keeps
55
+ * a caller that still builds its own `SystemMessage` working unchanged — the text simply
56
+ * travels a different route and lands in the same place. Only the leading run is taken:
57
+ * a system message deliberately placed mid-conversation is an operator instruction whose
58
+ * position carries meaning, and moving it would change what the model sees.
59
+ */
60
+ const takeLeadingSystem = (msgs: MessageFieldWithRole[]): string[] => {
61
+ const carried: string[] = []
62
+ while (msgs.length > 0 && isSystem(msgs[0])) {
63
+ const [msg] = msgs.splice(0, 1)
64
+ const text = textOf(msg.content)
65
+ if (text !== '') {
66
+ carried.push(text)
67
+ }
68
+ }
69
+
70
+ return carried
71
+ }
72
+
31
73
  /**
32
74
  * Build the four-method model API on top of a LangChain chat model.
33
75
  *
@@ -43,6 +85,9 @@ export const makeLlmModel = ({
43
85
  captureNull = false,
44
86
  retries = DEFAULT_MODEL_RETRIES,
45
87
  purpose,
88
+ prompt,
89
+ prompts,
90
+ files,
46
91
  }: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
47
92
 
48
93
  const ajv = new Ajv({ strict: false })
@@ -53,15 +98,67 @@ export const makeLlmModel = ({
53
98
  const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
54
99
  const timeout = idleTimeout(config)
55
100
 
56
- /** Normalize, then apply every in-place prompt adaptation, in dependency order. */
57
- const prepare = (input: ModelInput, useCache: boolean, cacheMax: number, json: boolean): MessageFieldWithRole[] => {
101
+ /**
102
+ * Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
103
+ * dependency order.
104
+ *
105
+ * With no prompt service wired this is exactly what it always was — the caller's
106
+ * messages, a JSON nudge, `/no_think`, cache markers. With one, the caller's leading
107
+ * system text is folded into a composed prompt whose stable sections come first, which
108
+ * is the whole point: a prompt cache is a PREFIX match, so the bytes every call shares
109
+ * have to be physically ahead of the bytes that differ.
110
+ */
111
+ const prepare = async (
112
+ input: ModelInput,
113
+ action: string,
114
+ useCache: boolean,
115
+ cacheMax: number,
116
+ json: boolean,
117
+ callSkills?: string[],
118
+ ): Promise<MessageFieldWithRole[]> => {
58
119
  const msgs = normalizeInput(input)
120
+ // The caller may hand back messages this pipeline marked on a PREVIOUS call — the
121
+ // markers live on its own objects. The budget is per request, so clear them and
122
+ // re-place our own below; otherwise they accumulate until the provider 400s.
123
+ stripCacheMarkers(msgs)
124
+ let reserved = 0
125
+
126
+ if (prompts != null) {
127
+ const carried = takeLeadingSystem(msgs)
128
+ const composed = await prompts().compose(
129
+ {
130
+ ...prompt,
131
+ context: [...(prompt?.context ?? []), ...carried],
132
+ callSkills: callSkills ?? prompt?.callSkills,
133
+ },
134
+ msgs,
135
+ { model, provider: plugin, purpose, action, cacheMax, files },
136
+ )
137
+ if (composed.system != null) {
138
+ msgs.unshift(composed.system)
139
+ reserved = composed.breakpoints
140
+ } else {
141
+ // Defensive: nothing was contributed, so hand the caller's own text straight back.
142
+ for (let i = carried.length - 1; i >= 0; i--) {
143
+ msgs.unshift({ role: 'system', content: carried[i] })
144
+ }
145
+ }
146
+ }
147
+
59
148
  if (json) ensureJsonMention(msgs)
60
149
  applyNoThink(msgs, config.disableThinking)
61
150
  // Cache markers replace string content with content blocks, so they must go last.
62
- if (plugin?.patchCache?.(msgs, { model, useCache, cacheMax }) === true) {
63
- console.log(`Prompt caching enabled for ${plugin.type} (up to ${cacheMax} breakpoints)`)
151
+ const ttl = prompt?.cacheTtl
152
+ const marked = plugin?.patchCache?.(msgs, {
153
+ model, useCache, cacheMax, reserved, ...(ttl != null ? { ttl } : {}),
154
+ })
155
+ if (reserved > 0 || marked === true) {
156
+ console.log(
157
+ `Prompt caching for ${plugin?.type}: ${reserved} system breakpoint(s)`
158
+ + `${marked === true ? ', 1 message breakpoint' : ''}`
159
+ )
64
160
  }
161
+
65
162
  return msgs
66
163
  }
67
164
 
@@ -197,8 +294,11 @@ export const makeLlmModel = ({
197
294
  }
198
295
 
199
296
  const helper: LlmModel = {
200
- ask: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmAskOptions) => {
201
- const msgs = prepare(input, useCache, cacheMax, false)
297
+ ask: async (
298
+ input,
299
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmAskOptions
300
+ ) => {
301
+ const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
202
302
  return withRetry({ retries, outputErrors }, async i => {
203
303
  const refined = refineModel(i)
204
304
  console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
@@ -242,8 +342,11 @@ export const makeLlmModel = ({
242
342
  })
243
343
  },
244
344
 
245
- talk: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmTalkOptions) => {
246
- const msgs = prepare(input, useCache, cacheMax, false)
345
+ talk: async (
346
+ input,
347
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmTalkOptions
348
+ ) => {
349
+ const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
247
350
  return withRetry({ retries, outputErrors }, async i => {
248
351
  const refined = refineModel(i)
249
352
  console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
@@ -277,9 +380,9 @@ export const makeLlmModel = ({
277
380
  invoke: async <T>(
278
381
  input: ModelInput,
279
382
  schema: JSONSchemaType<T>,
280
- { temperature, ref, filter, action, useCache = false, cacheMax = 4 }: LlmInvokeOptions<T>
383
+ { temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmInvokeOptions<T>
281
384
  ) => {
282
- const msgs = prepare(input, useCache, cacheMax, true)
385
+ const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
283
386
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
284
387
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
285
388
 
@@ -325,9 +428,9 @@ export const makeLlmModel = ({
325
428
  request: async <T>(
326
429
  input: ModelInput,
327
430
  schema: JSONSchemaType<T>,
328
- { ref, filter, action, useCache = false, cacheMax = 4 }: LlmRequestOptions
431
+ { ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills }: LlmRequestOptions
329
432
  ) => {
330
- const msgs = prepare(input, useCache, cacheMax, true)
433
+ const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
331
434
  const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
332
435
  const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
333
436