@owlmeans/llm 0.1.15 → 0.1.16-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/agent-meta/instructions/llm-prompt-caching.instructions.md +100 -0
  2. package/agent-meta/instructions/llm.instructions.md +13 -3
  3. package/agent-meta/manifest.json +16 -2
  4. package/agent-meta/skills/llm/SKILL.md +54 -4
  5. package/agent-meta/skills/llm-prompt-caching/SKILL.md +135 -0
  6. package/build/consts.d.ts +36 -3
  7. package/build/consts.d.ts.map +1 -1
  8. package/build/consts.js +37 -4
  9. package/build/consts.js.map +1 -1
  10. package/build/execution/service.d.ts.map +1 -1
  11. package/build/execution/service.js +8 -3
  12. package/build/execution/service.js.map +1 -1
  13. package/build/execution/types.d.ts +20 -1
  14. package/build/execution/types.d.ts.map +1 -1
  15. package/build/execution/utils.d.ts +12 -1
  16. package/build/execution/utils.d.ts.map +1 -1
  17. package/build/execution/utils.js +22 -0
  18. package/build/execution/utils.js.map +1 -1
  19. package/build/helpers/cache.d.ts +17 -0
  20. package/build/helpers/cache.d.ts.map +1 -0
  21. package/build/helpers/cache.js +22 -0
  22. package/build/helpers/cache.js.map +1 -0
  23. package/build/helpers/index.d.ts +1 -0
  24. package/build/helpers/index.d.ts.map +1 -1
  25. package/build/helpers/index.js +1 -0
  26. package/build/helpers/index.js.map +1 -1
  27. package/build/helpers/spectate.d.ts.map +1 -1
  28. package/build/helpers/spectate.js +7 -0
  29. package/build/helpers/spectate.js.map +1 -1
  30. package/build/index.d.ts +1 -0
  31. package/build/index.d.ts.map +1 -1
  32. package/build/index.js +1 -0
  33. package/build/index.js.map +1 -1
  34. package/build/model.d.ts +1 -1
  35. package/build/model.d.ts.map +1 -1
  36. package/build/model.js +90 -16
  37. package/build/model.js.map +1 -1
  38. package/build/plugins/anthropic.d.ts.map +1 -1
  39. package/build/plugins/anthropic.js +136 -21
  40. package/build/plugins/anthropic.js.map +1 -1
  41. package/build/plugins/openai.d.ts +9 -0
  42. package/build/plugins/openai.d.ts.map +1 -1
  43. package/build/plugins/openai.js +23 -1
  44. package/build/plugins/openai.js.map +1 -1
  45. package/build/plugins/types.d.ts +38 -5
  46. package/build/plugins/types.d.ts.map +1 -1
  47. package/build/prompt/index.d.ts +5 -0
  48. package/build/prompt/index.d.ts.map +1 -0
  49. package/build/prompt/index.js +4 -0
  50. package/build/prompt/index.js.map +1 -0
  51. package/build/prompt/plugins.d.ts +24 -0
  52. package/build/prompt/plugins.d.ts.map +1 -0
  53. package/build/prompt/plugins.js +64 -0
  54. package/build/prompt/plugins.js.map +1 -0
  55. package/build/prompt/render.d.ts +28 -0
  56. package/build/prompt/render.d.ts.map +1 -0
  57. package/build/prompt/render.js +39 -0
  58. package/build/prompt/render.js.map +1 -0
  59. package/build/prompt/service.d.ts +16 -0
  60. package/build/prompt/service.d.ts.map +1 -0
  61. package/build/prompt/service.js +145 -0
  62. package/build/prompt/service.js.map +1 -0
  63. package/build/prompt/types.d.ts +101 -0
  64. package/build/prompt/types.d.ts.map +1 -0
  65. package/build/prompt/types.js +2 -0
  66. package/build/prompt/types.js.map +1 -0
  67. package/build/service.d.ts.map +1 -1
  68. package/build/service.js +3 -0
  69. package/build/service.js.map +1 -1
  70. package/build/types.d.ts +53 -3
  71. package/build/types.d.ts.map +1 -1
  72. package/build/utils/prompt.d.ts +14 -0
  73. package/build/utils/prompt.d.ts.map +1 -1
  74. package/build/utils/prompt.js +32 -0
  75. package/build/utils/prompt.js.map +1 -1
  76. package/package.json +13 -6
  77. package/src/consts.ts +43 -5
  78. package/src/execution/service.ts +9 -3
  79. package/src/execution/types.ts +21 -2
  80. package/src/execution/utils.ts +28 -1
  81. package/src/helpers/cache.ts +33 -0
  82. package/src/helpers/index.ts +1 -0
  83. package/src/helpers/spectate.ts +10 -0
  84. package/src/index.ts +1 -0
  85. package/src/model.ts +119 -16
  86. package/src/plugins/anthropic.ts +159 -20
  87. package/src/plugins/openai.ts +26 -1
  88. package/src/plugins/types.ts +42 -5
  89. package/src/prompt/index.ts +5 -0
  90. package/src/prompt/plugins.ts +69 -0
  91. package/src/prompt/render.ts +48 -0
  92. package/src/prompt/service.ts +194 -0
  93. package/src/prompt/types.ts +114 -0
  94. package/src/service.ts +3 -0
  95. package/src/types.ts +54 -3
  96. package/src/utils/prompt.ts +33 -0
  97. package/tests/execution.spec.ts +42 -0
  98. package/tests/plugins.spec.ts +279 -14
  99. package/tests/prompt.spec.ts +194 -0
@@ -0,0 +1,100 @@
1
+ ---
2
+ description: "How the OwlMeans LLM layer composes a system prompt from a role and skills, and the prompt-cache rules that layout exists to satisfy — block order, breakpoint budget, determinism invariants, and the provider facts behind them. Consult before changing prompt composition, skills, a prompt plugin, or anything a request sends before its first per-call byte."
3
+ applyTo: "**/*.ts, **/*.tsx"
4
+ ---
5
+ <!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
6
+
7
+ # Prompt composition and caching
8
+
9
+ **Layer:** Core · **Packages:** `@owlmeans/llm` (`./prompt`), `@owlmeans/llm-common`, `@owlmeans/agent-skills` (`./llm`)
10
+
11
+ A prompt cache is an **exact prefix match**. Everything below follows from that: the bytes
12
+ every call shares must come first, physically, and must be identical down to the
13
+ whitespace.
14
+
15
+ ## The block layout
16
+
17
+ `PromptService.compose()` renders four ordered blocks into one system message:
18
+
19
+ | # | Block | Contributed by | Changes | Breakpoint |
20
+ |---|-------|----------------|---------|-----------|
21
+ | 0 | `Role` | `rolePlugin` ← `PromptPolicy.role` | per role | — |
22
+ | 1 | `Skills` | `skillsPlugin` ← registry + `inline` | per helper | ✅ closes role+skills |
23
+ | 2 | `Packages` | app plugins | per request | ✅ its own |
24
+ | 3 | `Context` | `contextPlugin` ← `context`, `callSkills` | per call | **never** (unless sole block) |
25
+
26
+ A caller's own leading `SystemMessage` is detached by `makeLlmModel` and re-emitted as
27
+ `Context`, so a helper that has not adopted `prompt: { role, skills }` keeps working.
28
+
29
+ ## Rules
30
+
31
+ - **Order is the cache key.** `PROMPT_BLOCK_ORDER` is explicit; skills sort by
32
+ `(order, alias)` with a code-unit comparison (never `localeCompare`); detected packages
33
+ sort alphabetically.
34
+ - **Skill bodies are pure constants** — no timestamps, paths, or interpolated request data.
35
+ - **A volatile block is never marked.** The trailing `Context` block changes every
36
+ call: a breakpoint there would pay a cache WRITE on every request and never read
37
+ one back. It is marked ONLY when it is the entire prompt (a caller that has not
38
+ adopted role/skills), because there it IS the stable part. Its parts are merged
39
+ into one chunk for the same reason — nothing downstream needs them separable.
40
+ - **Volatile content goes last.** Per-call skills use `LlmCallOptions.skills` (→ `Context`),
41
+ never the execution's `skills` (→ the cached `Skills` block).
42
+ - **Budget: 4 breakpoints per request.** The system prompt spends at most 3.
43
+ - **The last message is never cached** — it is the per-call payload.
44
+ - **Markers are placed in-place, on the caller's objects.** A caller that carries its
45
+ message array across calls (a coder's growing conversation, a fix loop) hands them back
46
+ still marked, and they accumulate. `prepare()` therefore calls `stripCacheMarkers()`
47
+ first, so the per-request count depends on THIS call alone. Anything that places a
48
+ marker outside that pipeline must do the same.
49
+ - **A prefix below `MIN_CACHEABLE_TOKENS` is left unmarked** (`ModelConfig.cacheMinTokens`
50
+ overrides per alias).
51
+ - **A prompt plugin MUST be deterministic**, and registering the same alias twice replaces
52
+ rather than appends.
53
+
54
+ ## Verifying it works
55
+ > **The smoke test cannot catch a caching bug.** It makes no model calls. A breakpoint-budget
56
+ > or marker-accumulation fault only appears under a real multi-call agent run — and it
57
+ > surfaces as a `400`, which the retry loop can bury for minutes. Check the agent's own pod
58
+ > logs for `Prompt cache [` lines and a non-zero `read`.
59
+
60
+
61
+ `readCacheUsage(message)` from `@owlmeans/llm/helpers` reads
62
+ `usage_metadata.input_token_details`. If `read` stays 0 across repeated calls that share a
63
+ prefix, something is invalidating it — diff `PromptResult.blocks` between two calls.
64
+
65
+ ## Package skills in a prompt (`@owlmeans/agent-skills/llm`)
66
+
67
+ `owlmeansPackagesPlugin(options)` notices which `@owlmeans/*` packages a request mentions
68
+ and loads their published skills into the `Packages` block. Everything a package documents
69
+ is loaded — there is no relevance filtering, by design.
70
+
71
+ Resolution order per package: the host's `LlmFileProvider` (the only path that sees a
72
+ sandbox or remote workspace) → an installed copy under `node_modules` → the canonical
73
+ repository over HTTPS. Every failure is a miss, never a throw. Results, including misses,
74
+ are cached per plugin instance.
75
+
76
+ ```typescript
77
+ ctx.prompts().use(owlmeansPackagesPlugin({
78
+ files: () => ctx.files(), // tried first
79
+ exclude: ['@owlmeans/llm'], // already covered by the static Skills block
80
+ fetch: false, // air-gapped: skip the repository fallback
81
+ }))
82
+ ```
83
+
84
+ The manifest deliberately carries no git ref — version-matching comes from shipping the
85
+ copy inside the tarball, so for a package that is NOT installed the ref is a plugin option
86
+ (`ref`, default `main`).
87
+
88
+ ## Provider facts these rules encode
89
+
90
+ **Anthropic** ([docs](https://platform.claude.com/docs/en/build-with-claude/prompt-caching.md)):
91
+ render order `tools` → `system` → `messages`; max 4 breakpoints; minimum cacheable prefix is
92
+ model-dependent and **not monotonic** (512 newest / 1024 most / 4096 on Opus 4.6-4.5 and
93
+ Haiku 4.5); writes cost ~1.25× at 5m TTL and ~2× at 1h; a `system` change invalidates
94
+ system + messages but not `tools`.
95
+
96
+ **OpenAI** ([docs](https://developers.openai.com/api/docs/guides/prompt-caching)): automatic
97
+ from 1024 tokens; routing hashes roughly the first 256 tokens and mixes in
98
+ `prompt_cache_key`. The `openai` plugin sets it from the config alias; the `compatible`
99
+ plugin does not, because aggregators with `provider.require_parameters` can drop providers
100
+ over an unknown field.
@@ -1,5 +1,5 @@
1
1
  ---
2
- description: "How to use @owlmeans/llm — the LLM inference runtime: four-method model, provider plugins, model factory service, and the policy-driven execution abstraction."
2
+ description: "How to use @owlmeans/llm — the LLM inference runtime: four-method model, provider plugins, model factory service, the policy-driven execution abstraction, and the prompt/skill composition service."
3
3
  applyTo: "**/*.ts, **/*.tsx"
4
4
  ---
5
5
  <!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
@@ -7,7 +7,7 @@ applyTo: "**/*.ts, **/*.tsx"
7
7
  # @owlmeans/llm
8
8
 
9
9
  **Layer:** Core
10
- **Install:** `"@owlmeans/llm": "^0.1.15"` in `dependencies` (plus the `@langchain/*` peers)
10
+ **Install:** `"@owlmeans/llm": "^0.1.16-rc.0"` in `dependencies` (plus the `@langchain/*` peers)
11
11
 
12
12
  The inference runtime. Everything provider-specific is a **plugin**; the model owns only the
13
13
  provider-independent parts. Serializable contracts live in `@owlmeans/llm-common`.
@@ -19,6 +19,8 @@ provider-independent parts. Serializable contracts live in `@owlmeans/llm-common
19
19
  | `makeLlmModel(options, spectator)` | `ask` / `talk` / `invoke` / `request`. |
20
20
  | `makeLlmService` · `appendLlmService` · `llmServiceApi` | Model factory/registry; `llmServiceApi` composes into your own service. |
21
21
  | `makeExecutionService` · `appendExecutionService` · `executionServiceApi` | Frozen 3-level executions, policy resolution, snapshot/restore/checkpoint. |
22
+ | `makePromptService` · `appendPromptService` · `promptServiceApi` | Skill registry + composition plugin chain (`@owlmeans/llm/prompt`). |
23
+ | `renderSkill`, `sortSkills`, `compareAlias`, `readCacheUsage` | Deterministic rendering and prompt-cache accounting. |
22
24
  | `plugins`, `registerLlmPlugin`, `resolvePlugin`, `pluginOf`, `pluginFor` | Provider-plugin registry (`@owlmeans/llm/plugins`). |
23
25
  | `anthropicPlugin`, `openAiPlugin`, `compatiblePlugin`, `openAiFamily` | Built-ins; spread `openAiFamily` into a new OpenAI-compatible plugin. |
24
26
  | `withRetry`, `registerFatalError`, `spectate`, `normalizeInput`, `parseJsonContent`, `coerceToSchema` | Helpers (`@owlmeans/llm/helpers`). |
@@ -30,14 +32,22 @@ provider-independent parts. Serializable contracts live in `@owlmeans/llm-common
30
32
  exported. Place a new function on the right side rather than exporting a util for a test.
31
33
  - Never branch on the provider (`instanceof ChatAnthropic`, `provider === …`) in the model or
32
34
  the service — put it on the `LlmPlugin` (`build` / `owns` / `family` / `refine` /
33
- `structuredMode` / `toolChoice` / `responseFormat` / `patchCache` / `isFatal`).
35
+ `structuredMode` / `toolChoice` / `responseFormat` / `patchSystem` / `patchCache` / `isFatal`).
34
36
  - Plugin registration order is load-bearing: `pluginFor` returns the first `owns` match, and
35
37
  `compatible` precedes `openai` so an unlabelled `ChatOpenAI` gets the conservative
36
38
  tool-calling behaviour.
39
+ - Never `instanceof` an error from a provider SDK: `@langchain/*` bundle their own
40
+ nested copies, so the check silently fails and a fatal `400` gets retried eight
41
+ times. Match `status === 400` instead.
37
42
  - `@langchain/*` are **peer** dependencies — model instances cross the package boundary and
38
43
  two installed copies are nominally distinct types. Pin one copy in the consumer.
39
44
  - Extend the execution generically: your own `ExecutionShape`, your collaborator fields in
40
45
  `collaboratorKeys`. Do not narrow inherited method signatures (contravariance error).
46
+ - Never hand-build a persona as a `SystemMessage` in a helper. Declare it as
47
+ `PromptPolicy.role` plus registered skills and let the prompt service compose it —
48
+ otherwise the knowledge duplicates and the cacheable prefix stops being stable. Skills
49
+ accumulate project → task → helper; the deepest declared role wins (`mergePrompt`).
50
+ Block order, breakpoint budget and the provider facts: `llm-prompt-caching`.
41
51
 
42
52
  ## Usage
43
53
 
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "schemaVersion": 1,
3
3
  "package": "@owlmeans/llm",
4
- "version": "0.1.15",
5
- "generatedAt": "2026-08-07T17:06:32.901Z",
4
+ "version": "0.1.16-rc.0",
5
+ "generatedAt": "2026-08-11T14:02:34.979Z",
6
6
  "canonicalRepo": "https://github.com/owlmeans/common",
7
7
  "entries": [
8
8
  {
@@ -12,12 +12,26 @@
12
12
  "file": "skills/llm/SKILL.md",
13
13
  "canonicalPath": ".claude/skills/llm/SKILL.md"
14
14
  },
15
+ {
16
+ "kind": "skill",
17
+ "name": "llm-prompt-caching",
18
+ "category": "multi-package",
19
+ "file": "skills/llm-prompt-caching/SKILL.md",
20
+ "canonicalPath": ".claude/skills/llm-prompt-caching/SKILL.md"
21
+ },
15
22
  {
16
23
  "kind": "instruction",
17
24
  "name": "llm",
18
25
  "category": "package-specific",
19
26
  "file": "instructions/llm.instructions.md",
20
27
  "canonicalPath": ".github/instructions/llm.instructions.md"
28
+ },
29
+ {
30
+ "kind": "instruction",
31
+ "name": "llm-prompt-caching",
32
+ "category": "multi-package",
33
+ "file": "instructions/llm-prompt-caching.instructions.md",
34
+ "canonicalPath": ".github/instructions/llm-prompt-caching.instructions.md"
21
35
  }
22
36
  ]
23
37
  }
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: llm
3
- description: How to use @owlmeans/llm — the LLM inference runtime (four-method model, provider plugins, model factory service, policy-driven execution abstraction). Auto-invoked when importing the model, an LlmPlugin, the LlmService, or the execution service.
3
+ description: How to use @owlmeans/llm — the LLM inference runtime (four-method model, provider plugins, model factory service, policy-driven execution abstraction, and the prompt/skill composition service). Auto-invoked when importing the model, an LlmPlugin, the LlmService, the execution service, or the prompt service.
4
4
  user-invocable: false
5
5
  ---
6
6
  <!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
@@ -8,7 +8,7 @@ user-invocable: false
8
8
  # @owlmeans/llm
9
9
 
10
10
  **Layer:** Core
11
- **Install:** `"@owlmeans/llm": "^0.1.15"` in `dependencies` (plus the `@langchain/*` peers)
11
+ **Install:** `"@owlmeans/llm": "^0.1.16-rc.0"` in `dependencies` (plus the `@langchain/*` peers)
12
12
 
13
13
  The inference runtime. Everything provider-specific is a **plugin**; the model itself only
14
14
  owns the provider-independent parts (streaming discipline, retries, validation,
@@ -22,12 +22,17 @@ observability). Serializable contracts live in `@owlmeans/llm-common`.
22
22
  | `makeLlmService(options, alias?)` · `appendLlmService(ctx, options, alias?)` | Model factory/registry — resolves a `ModelConfig` by alias, memoized per alias+override. |
23
23
  | `llmServiceApi(options, self)` | The factory half WITHOUT `createService`, to compose into your own service (role accessors, domain helpers). |
24
24
  | `makeExecutionService(alias?, options?)` · `appendExecutionService(ctx, alias?, options?)` | Frozen 3-level executions + policy resolution + snapshot/restore/checkpoint. |
25
+ | `makePromptService(options?, alias?)` · `appendPromptService(ctx, options?, alias?)` · `promptServiceApi(options, self)` | Skill registry + the composition plugin chain. Also at `@owlmeans/llm/prompt`. |
26
+ | `rolePlugin`, `skillsPlugin`, `contextPlugin`, `BUILT_IN_PROMPT_PLUGINS` | The built-in composition plugins. |
27
+ | `renderSkill`, `sortSkills`, `joinChunks`, `compareAlias`, `prefixHash` | Deterministic rendering primitives — reuse them, never re-implement. |
28
+ | `readCacheUsage`, `hasCacheActivity` | Normalized prompt-cache accounting from a completion. |
25
29
  | `executionServiceApi(options, self)` | The execution half without `createService`, for the same composition pattern. |
26
30
  | `plugins`, `registerLlmPlugin`, `resolvePlugin`, `pluginOf`, `pluginFor` | The provider-plugin registry. Also at `@owlmeans/llm/plugins`. |
27
31
  | `anthropicPlugin`, `openAiPlugin`, `compatiblePlugin`, `openAiFamily` | Built-in providers; `openAiFamily` is the shared OpenAI-client behaviour to spread into a new plugin. |
28
32
  | `withRetry`, `registerFatalError`, `spectate`, `normalizeInput`, `parseJsonContent`, `coerceToSchema` | Helpers usable alongside a model. Also at `@owlmeans/llm/helpers`. |
29
33
  | `LlmError`, `LlmModelError`, `LlmMissconfiguredError`, `LlmPluginError`, `LlmRetryExceededError` | `ResilientError` family. `LlmModelError` is the RETRYABLE one. |
30
- | `DEFAULT_MODEL_RETRIES`, `MODEL_STREAM_TIMEOUT_MS`, `FALLBACK_AFTER_ATTEMPTS`, `DEFAULT_EFFORT`, `EFFORT_TABLE`, `LLM_SERVICE`, `EXECUTION_SERVICE` | Tuning + aliases. |
34
+ | `mergePrompt`, `mergePolicy`, `resolveRole`, `effortPatch` | Execution merge helpers; `mergePrompt` unions skills and takes the deepest role. |
35
+ | `DEFAULT_MODEL_RETRIES`, `MODEL_STREAM_TIMEOUT_MS` (3 min idle), `FALLBACK_AFTER_ATTEMPTS`, `DEFAULT_EFFORT`, `EFFORT_TABLE`, `MAX_CACHE_BREAKPOINTS`, `MAX_SYSTEM_BREAKPOINTS`, `MIN_CACHEABLE_TOKENS`, `LLM_SERVICE`, `EXECUTION_SERVICE`, `PROMPT_SERVICE` | Tuning + aliases. |
31
36
 
32
37
  ## helpers/ vs utils/ — the rule this package follows
33
38
 
@@ -50,7 +55,9 @@ Adding a function? Decide which side it belongs to first, then place it. Do not
50
55
  | `refine` | the per-provider retry rebuild (budget doubling, reasoning shrink) |
51
56
  | `structuredMode` | native `response_format` vs the forced-tool-call hack |
52
57
  | `toolChoice` / `responseFormat` | the provider-specific call shapes |
53
- | `patchCache` | prompt-cache markers |
58
+ | `patchSystem` | how the composed system blocks are rendered and where their cache breakpoints go |
59
+ | `patchCache` | the message-prefix cache marker |
60
+ | `cacheKey` (via `ModelConfig.cacheKey`) | provider cache-routing hints such as OpenAI's `prompt_cache_key` |
54
61
  | `isFatal` | "this error can never be retried" |
55
62
 
56
63
  **Registration order is load-bearing.** Instance-based lookup (`pluginFor`) returns the
@@ -58,6 +65,19 @@ FIRST plugin whose `owns` matches. `compatible` is registered before `openai` be
58
65
  build a `ChatOpenAI`, and assuming the tool-calling hack for an unlabelled model is safe
59
66
  everywhere while assuming native JSON-schema support is not.
60
67
 
68
+ ## System prompts: a role and skills, never a hand-built message
69
+
70
+ `makeLlmModel` takes `prompt` (a `PromptInput`) and a `prompts` resolver. The prompt service
71
+ composes them into an ordered, cacheable system message; a caller's own leading
72
+ `SystemMessage` is folded into the volatile `Context` block, so an unmigrated call site
73
+ still works. **Do not build a persona as a `SystemMessage` in a helper** — declare it as
74
+ `PromptPolicy.role` plus registered skills, or the knowledge duplicates and the cache
75
+ prefix stops being stable.
76
+
77
+ Skills accumulate down the execution chain (project → task → helper) and the deepest
78
+ declared `role` wins — see `mergePrompt`. Full rules, breakpoint budget and the provider
79
+ facts behind them: [[llm-prompt-caching]].
80
+
61
81
  ## Execution: policy in, model out
62
82
 
63
83
  ```
@@ -66,6 +86,9 @@ ProjectExecution ← root: policy + purpose + models resolver
66
86
  └─ HelperExecution ← + a RESOLVED model + temperatureFactory, bound to a role
67
87
  ```
68
88
 
89
+ `prompt` (role + skills) travels on `ExecutionState`, so it survives snapshot/restore;
90
+ `prompts` and `files` are collaborators and never enter a snapshot.
91
+
69
92
  Every method returns a NEW `Object.freeze`d object. Resolution precedence in
70
93
  `model(exec, role, override)`: **roleOverride → modelOverride → effort tier →
71
94
  `LlmService.getModel`**. `escalate(exec, { effort })` raises the tier once and cascades to
@@ -79,6 +102,19 @@ inherited method signatures**, which would be a contravariance error.
79
102
  `snapshot` excludes `state` itself; without that, every `derive`/`escalate`/`withPurpose`
80
103
  on a task would nest another copy of the previous state.
81
104
 
105
+ ## Never `instanceof` an error from a provider SDK
106
+
107
+ `@langchain/anthropic` and `@langchain/openai` bundle their OWN nested copies of the
108
+ provider SDKs, so an error they throw is an instance of a DIFFERENT class than the one this
109
+ package imports — `e instanceof BadRequestError` silently returns `false`. `isFatal` used
110
+ that check, so every fatal `400` was classified retryable and burned all eight attempts,
111
+ turning one malformed request into minutes of thrash with the real message buried under the
112
+ retries.
113
+
114
+ Match on the wire shape instead — `(e as { status?: unknown })?.status === 400`. The same
115
+ trap applies to any cross-copy `instanceof`; it is the runtime face of the peer-dependency
116
+ identity rule below.
117
+
82
118
  ## Peer-dependency rule (langchain identity)
83
119
 
84
120
  `@langchain/core`, `@langchain/openai` and `@langchain/anthropic` are **peer** dependencies:
@@ -99,6 +135,19 @@ const model = makeLlmModel(
99
135
  const spec = await model.invoke('Describe the app', SpecSchema, { action: 'spec' })
100
136
  ```
101
137
 
138
+ ## Hangs are bounded by an IDLE deadline, not a total one
139
+
140
+ A stalled provider is aborted after `MODEL_STREAM_TIMEOUT_MS` (3 min) of SILENCE and
141
+ surfaces as a retryable `LlmModelError`, so the escalator moves on. The timer re-arms on
142
+ every token, so a long-but-productive generation is never cut off — which is why the value
143
+ can be low. Set it for a whole deployment with `LlmServiceOptions.streamTimeout` where the
144
+ application composes its context; a preset that names its own `ModelConfig.streamTimeout`
145
+ keeps it.
146
+
147
+ Note what this does NOT bound: a call that keeps streaming forever, and retries. A fatal
148
+ error misclassified as retryable multiplies its own latency by `DEFAULT_MODEL_RETRIES` —
149
+ see the `instanceof` trap above.
150
+
102
151
  ## Resilience already handled — do not reimplement
103
152
 
104
153
  Idle stream deadline · duplicate-final-chunk dedup · output-budget escalation · reasoning-cap
@@ -118,4 +167,5 @@ on `OPENROUTER_SECRET` / `ANTHROPIC_SECRET` in the repo-root `.env` and self-ski
118
167
  ## Related
119
168
 
120
169
  - [[llm-common]] — the serializable contracts
170
+ - [[llm-prompt-caching]] — prompt composition, block order and the cache invariants
121
171
  - [[context]] — service registration · [[error]] — the `ResilientError` family
@@ -0,0 +1,135 @@
1
+ ---
2
+ name: llm-prompt-caching
3
+ description: How the OwlMeans LLM layer composes a system prompt from a role and skills, and the prompt-cache rules that layout exists to satisfy — block order, breakpoint budget, determinism invariants, and the provider facts behind them. Auto-invoked when touching prompt composition, skills, LlmPromptPlugin, patchSystem/patchCache, or anything that changes what a request sends before its first per-call byte.
4
+ user-invocable: false
5
+ ---
6
+ <!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
7
+
8
+ # Prompt composition and caching
9
+
10
+ **Layer:** Core · **Packages:** `@owlmeans/llm` (`./prompt`), `@owlmeans/llm-common`, `@owlmeans/agent-skills` (`./llm`)
11
+
12
+ A prompt cache is an **exact prefix match**. Everything in this document follows from that
13
+ one fact: the bytes every call shares must come first, physically, and must be identical
14
+ down to the whitespace. The block layout, the ordering rules and the determinism
15
+ invariants are not style — they are the cache.
16
+
17
+ ## The block layout
18
+
19
+ `PromptService.compose()` renders four ordered blocks (`PromptBlock`, `PROMPT_BLOCK_ORDER`)
20
+ into a single system message:
21
+
22
+ | # | Block | Contributed by | Changes | Breakpoint |
23
+ |---|-------|----------------|---------|-----------|
24
+ | 0 | `Role` | `rolePlugin` ← `PromptPolicy.role` | per role | — |
25
+ | 1 | `Skills` | `skillsPlugin` ← registry + `inline` | per helper | ✅ closes role+skills |
26
+ | 2 | `Packages` | app plugins (e.g. `owlmeansPackagesPlugin`) | per request | ✅ its own |
27
+ | 3 | `Context` | `contextPlugin` ← `context`, `callSkills` | per call | **never** (unless sole block) |
28
+
29
+ A caller's own leading `SystemMessage` is detached by `makeLlmModel` and re-emitted as
30
+ `Context`, so a helper that has not adopted `prompt: { role, skills }` keeps working —
31
+ its text simply travels a different route.
32
+
33
+ ## Rules
34
+
35
+ - **Order is the cache key.** `PROMPT_BLOCK_ORDER` is declared explicitly, skills sort by
36
+ `(order, alias)` with a code-unit comparison (`compareAlias`, never `localeCompare` —
37
+ ICU data differs between hosts), and detected packages sort alphabetically.
38
+ - **Skill bodies are pure constants.** No timestamps, no absolute paths, no interpolated
39
+ request data. One varying byte invalidates the prefix for every call that shares it.
40
+ - **A volatile block is never marked.** The trailing `Context` block changes every
41
+ call: a breakpoint there would pay a cache WRITE on every request and never read
42
+ one back. It is marked ONLY when it is the entire prompt (a caller that has not
43
+ adopted role/skills), because there it IS the stable part. Its parts are merged
44
+ into one chunk for the same reason — nothing downstream needs them separable.
45
+ - **Volatile content goes last.** Per-call skills belong in `LlmCallOptions.skills`
46
+ (→ `Context`), not in the execution's `skills` (→ the cached `Skills` block).
47
+ - **Budget: 4 breakpoints per request, total** — across tools, system AND messages.
48
+ Anthropic rejects the fifth outright (`400 A maximum of 4 blocks with cache_control may
49
+ be provided. Found 5.`), and a 400 is fatal, so the whole call fails. The system prompt
50
+ may spend at most `MAX_SYSTEM_BREAKPOINTS` (2) — its only STABLE boundaries are the end
51
+ of role+skills and the end of packages — which always leaves two for the messages.
52
+ - **The last message is never cached.** `patchCache` stops one short of the end: the final
53
+ message is the per-call payload, and `ensureJsonMention` / `applyNoThink` append to it.
54
+ - **Markers are placed in-place, on the caller's objects.** A caller that carries its
55
+ message array across calls (a coder's growing conversation, a fix loop) hands them back
56
+ still marked, and they accumulate. `prepare()` therefore calls `stripCacheMarkers()`
57
+ first, so the per-request count depends on THIS call alone. Anything that places a
58
+ marker outside that pipeline must do the same.
59
+ - **A short prefix is not marked.** Below `MIN_CACHEABLE_TOKENS` (override per alias with
60
+ `ModelConfig.cacheMinTokens`) a marker buys nothing and costs a breakpoint.
61
+
62
+ ## Verifying it works
63
+ > **The smoke test cannot catch a caching bug.** It makes no model calls. A breakpoint-budget
64
+ > or marker-accumulation fault only appears under a real multi-call agent run — and it
65
+ > surfaces as a `400`, which the retry loop can bury for minutes. Check the agent's own pod
66
+ > logs for `Prompt cache [` lines and a non-zero `read`.
67
+
68
+
69
+ `usage_metadata.input_token_details` is the only honest answer. `readCacheUsage(message)`
70
+ (`@owlmeans/llm/helpers`) normalizes it; `spectate` logs a line whenever a provider reports
71
+ any cache activity. **If `read` stays 0 across repeated calls that share a prefix, something
72
+ is invalidating it** — diff `PromptResult.blocks` between two calls to find what.
73
+
74
+ ## Adding a plugin
75
+
76
+ ```typescript
77
+ ctx.prompts().use({
78
+ alias: 'my-plugin',
79
+ order: 50, // built-ins hold 0 (role), 10 (skills), 90 (context)
80
+ compose: ctx => ctx.add(PromptBlock.Skills, text), // static, runs first
81
+ inspect: ctx => { /* reads ctx.messages */ }, // after every compose pass
82
+ })
83
+ ```
84
+
85
+ A plugin MUST be deterministic. Anything it contributes to a cached block and cannot
86
+ reproduce byte-for-byte breaks the prefix for everyone sharing it. Register by alias —
87
+ re-registering the same alias replaces rather than appends, so double-wiring cannot
88
+ double-emit.
89
+
90
+ ## Package skills in a prompt (`@owlmeans/agent-skills/llm`)
91
+
92
+ `owlmeansPackagesPlugin(options)` notices which `@owlmeans/*` packages a request mentions
93
+ and loads their published skills into the `Packages` block. Everything a package documents
94
+ is loaded — there is no relevance filtering, by design.
95
+
96
+ Resolution order per package: the host's `LlmFileProvider` (the only path that sees a
97
+ sandbox or remote workspace) → an installed copy under `node_modules` → the canonical
98
+ repository over HTTPS. Every failure is a miss, never a throw. Results, including misses,
99
+ are cached per plugin instance.
100
+
101
+ ```typescript
102
+ ctx.prompts().use(owlmeansPackagesPlugin({
103
+ files: () => ctx.files(), // tried first
104
+ exclude: ['@owlmeans/llm'], // already covered by the static Skills block
105
+ fetch: false, // air-gapped: skip the repository fallback
106
+ }))
107
+ ```
108
+
109
+ The manifest deliberately carries no git ref — version-matching comes from shipping the
110
+ copy inside the tarball, so for a package that is NOT installed the ref is a plugin option
111
+ (`ref`, default `main`).
112
+
113
+ ## Provider facts these rules encode
114
+
115
+ **Anthropic** ([prompt-caching](https://platform.claude.com/docs/en/build-with-claude/prompt-caching.md)) —
116
+ render order is `tools` → `system` → `messages`; **max 4** `cache_control` breakpoints per
117
+ request; the minimum cacheable prefix is model-dependent and **not monotonic** across
118
+ generations (512 tokens on the newest models, 1024 on most, 4096 on Opus 4.6/4.5 and
119
+ Haiku 4.5); a write costs ~1.25× at the 5-minute TTL and ~2× at `ttl: '1h'`, so the long
120
+ TTL only pays from the third read; changing `system` invalidates system + messages but not
121
+ `tools`; each breakpoint looks back at most 20 content blocks.
122
+
123
+ **OpenAI** ([prompt-caching](https://developers.openai.com/api/docs/guides/prompt-caching)) —
124
+ caching is automatic from 1024 tokens in 128-token increments, with no markers to place.
125
+ Requests are routed by a hash of roughly the first 256 tokens, and `prompt_cache_key` is
126
+ mixed into that hash, so requests sharing a key land on the same backend and can actually
127
+ hit each other's entries. The `openai` plugin sets it from the config alias (one value per
128
+ role); the `compatible` plugin deliberately does not, because aggregators running with
129
+ `provider.require_parameters` can drop every serving provider over an unknown field.
130
+
131
+ ## Related
132
+
133
+ - [[llm]] — the runtime and the provider-plugin seam
134
+ - [[llm-common]] — `SkillDefinition`, `PromptPolicy`, `PromptBlock`, `LlmFileProvider`
135
+ - [[agent-skills]] — `@owlmeans/agent-skills/llm`, the package-skills plugin
package/build/consts.d.ts CHANGED
@@ -1,9 +1,11 @@
1
1
  import { ExecutionEffort } from '@owlmeans/llm-common';
2
- import type { ModelConfigPatch } from '@owlmeans/llm-common';
2
+ import type { CacheTtl, ModelConfigPatch } from '@owlmeans/llm-common';
3
3
  /** Context-service alias for the {@link LlmService} (model factory / registry). */
4
4
  export declare const LLM_SERVICE = "owlmeans-llm-service";
5
5
  /** Context-service alias for the {@link ExecutionService}. */
6
6
  export declare const EXECUTION_SERVICE = "owlmeans-llm-execution-service";
7
+ /** Context-service alias for the {@link PromptService} (skill registry + composition). */
8
+ export declare const PROMPT_SERVICE = "owlmeans-llm-prompt-service";
7
9
  /** Default number of attempts a single model call makes before giving up. */
8
10
  export declare const DEFAULT_MODEL_RETRIES = 8;
9
11
  /**
@@ -13,7 +15,13 @@ export declare const DEFAULT_MODEL_RETRIES = 8;
13
15
  * against a provider that accepts the request and then never streams anything (observed
14
16
  * with throughput-sorted OpenRouter routing), which would otherwise block forever —
15
17
  * `maxRetries` never helps there because the request never errors, it just hangs.
16
- * Overridable per model via `ModelConfig.streamTimeout`.
18
+ * Overridable per model via `ModelConfig.streamTimeout`, or for every model at once via
19
+ * `LlmServiceOptions.streamTimeout` where the application composes its context.
20
+ *
21
+ * Three minutes, not longer: the timer resets on every token, so a legitimately long
22
+ * generation is never at risk — this only bounds SILENCE. The cost of a high value is
23
+ * paid entirely by hung calls, and a hang that takes five minutes to notice is a hang
24
+ * that can stall an agent run for half an hour once retries multiply it.
17
25
  */
18
26
  export declare const MODEL_STREAM_TIMEOUT_MS: number;
19
27
  /**
@@ -30,8 +38,33 @@ export declare const FALLBACK_AFTER_ATTEMPTS = 3;
30
38
  * issue a 400 "max_tokens exceeds the model's per-request limit".
31
39
  */
32
40
  export declare const DEFAULT_MAX_OUTPUT_CAP: number;
33
- /** Provider hard limit on prompt-cache breakpoints (Anthropic). */
41
+ /** Provider hard limit on prompt-cache breakpoints per REQUEST (Anthropic). */
34
42
  export declare const MAX_CACHE_BREAKPOINTS = 4;
43
+ /**
44
+ * Share of {@link MAX_CACHE_BREAKPOINTS} the composed system prompt may spend.
45
+ *
46
+ * Two is all it can use: the only STABLE boundaries are the end of role+skills and the end
47
+ * of the packages block. The trailing context block is volatile and deliberately never
48
+ * marked, so the remaining two breakpoints always stay available to the message prefix.
49
+ */
50
+ export declare const MAX_SYSTEM_BREAKPOINTS = 2;
51
+ /** Cache lifetime used when neither the call nor the service asks for another. */
52
+ export declare const DEFAULT_CACHE_TTL: CacheTtl;
53
+ /**
54
+ * Smallest prefix worth marking as cacheable, in tokens. Providers silently refuse to
55
+ * create an entry below their own threshold (1024 tokens on most Claude models, 512 on
56
+ * the newest, 4096 on a few older ones — it is NOT monotonic across generations), so a
57
+ * marker on a short prefix costs nothing but wastes a breakpoint and produces a
58
+ * "caching enabled" log for an entry that was never written. Override per model with
59
+ * `ModelConfig.cacheMinTokens`.
60
+ */
61
+ export declare const MIN_CACHEABLE_TOKENS = 1024;
62
+ /**
63
+ * Characters per token used to size a prefix against {@link MIN_CACHEABLE_TOKENS}. A rough
64
+ * average for English prose and TypeScript; only ever used to decide whether marking is
65
+ * worth a breakpoint, never for billing or budgeting.
66
+ */
67
+ export declare const CHARS_PER_TOKEN = 4;
35
68
  /**
36
69
  * Appended to the prompt of `invoke`/`request` when no message already mentions JSON.
37
70
  * Some providers refuse or ignore JSON modes unless the word appears in the prompt;
@@ -1 +1 @@
1
- {"version":3,"file":"consts.d.ts","sourceRoot":"","sources":["../src/consts.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,eAAe,EAAE,MAAM,sBAAsB,CAAA;AACtD,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,sBAAsB,CAAA;AAE5D,mFAAmF;AACnF,eAAO,MAAM,WAAW,yBAAyB,CAAA;AAEjD,8DAA8D;AAC9D,eAAO,MAAM,iBAAiB,mCAAmC,CAAA;AAEjE,6EAA6E;AAC7E,eAAO,MAAM,qBAAqB,IAAI,CAAA;AAEtC;;;;;;;;GAQG;AACH,eAAO,MAAM,uBAAuB,QAAgB,CAAA;AAEpD;;;;;GAKG;AACH,eAAO,MAAM,uBAAuB,IAAI,CAAA;AAExC;;;;;GAKG;AACH,eAAO,MAAM,sBAAsB,QAAY,CAAA;AAE/C,mEAAmE;AACnE,eAAO,MAAM,qBAAqB,IAAI,CAAA;AAEtC;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,QAGkD,CAAA;AAE/E;;;;GAIG;AACH,eAAO,MAAM,kBAAkB,cAAc,CAAA;AAE7C,uFAAuF;AACvF,eAAO,MAAM,iBAAiB,YAAY,CAAA;AAE1C,8DAA8D;AAC9D,eAAO,MAAM,cAAc,2BAA2B,CAAA;AAEtD;;;;GAIG;AACH,eAAO,MAAM,YAAY,EAAE,MAAM,CAAC,eAAe,EAAE,gBAAgB,CAKlE,CAAA;AAED;;;;;;;GAOG;AACH,eAAO,MAAM,iBAAiB,EAAE,MAAM,EAErC,CAAA"}
1
+ {"version":3,"file":"consts.d.ts","sourceRoot":"","sources":["../src/consts.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,eAAe,EAAE,MAAM,sBAAsB,CAAA;AACtD,OAAO,KAAK,EAAE,QAAQ,EAAE,gBAAgB,EAAE,MAAM,sBAAsB,CAAA;AAEtE,mFAAmF;AACnF,eAAO,MAAM,WAAW,yBAAyB,CAAA;AAEjD,8DAA8D;AAC9D,eAAO,MAAM,iBAAiB,mCAAmC,CAAA;AAEjE,0FAA0F;AAC1F,eAAO,MAAM,cAAc,gCAAgC,CAAA;AAE3D,6EAA6E;AAC7E,eAAO,MAAM,qBAAqB,IAAI,CAAA;AAEtC;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,uBAAuB,QAAgB,CAAA;AAEpD;;;;;GAKG;AACH,eAAO,MAAM,uBAAuB,IAAI,CAAA;AAExC;;;;;GAKG;AACH,eAAO,MAAM,sBAAsB,QAAY,CAAA;AAE/C,+EAA+E;AAC/E,eAAO,MAAM,qBAAqB,IAAI,CAAA;AAEtC;;;;;;GAMG;AACH,eAAO,MAAM,sBAAsB,IAAI,CAAA;AAEvC,kFAAkF;AAClF,eAAO,MAAM,iBAAiB,EAAE,QAAe,CAAA;AAE/C;;;;;;;GAOG;AACH,eAAO,MAAM,oBAAoB,OAAO,CAAA;AAExC;;;;GAIG;AACH,eAAO,MAAM,eAAe,IAAI,CAAA;AAEhC;;;;;GAKG;AACH,eAAO,MAAM,gBAAgB,QAGkD,CAAA;AAE/E;;;;GAIG;AACH,eAAO,MAAM,kBAAkB,cAAc,CAAA;AAE7C,uFAAuF;AACvF,eAAO,MAAM,iBAAiB,YAAY,CAAA;AAE1C,8DAA8D;AAC9D,eAAO,MAAM,cAAc,2BAA2B,CAAA;AAEtD;;;;GAIG;AACH,eAAO,MAAM,YAAY,EAAE,MAAM,CAAC,eAAe,EAAE,gBAAgB,CAKlE,CAAA;AAED;;;;;;;GAOG;AACH,eAAO,MAAM,iBAAiB,EAAE,MAAM,EAErC,CAAA"}
package/build/consts.js CHANGED
@@ -3,6 +3,8 @@ import { ExecutionEffort } from '@owlmeans/llm-common';
3
3
  export const LLM_SERVICE = 'owlmeans-llm-service';
4
4
  /** Context-service alias for the {@link ExecutionService}. */
5
5
  export const EXECUTION_SERVICE = 'owlmeans-llm-execution-service';
6
+ /** Context-service alias for the {@link PromptService} (skill registry + composition). */
7
+ export const PROMPT_SERVICE = 'owlmeans-llm-prompt-service';
6
8
  /** Default number of attempts a single model call makes before giving up. */
7
9
  export const DEFAULT_MODEL_RETRIES = 8;
8
10
  /**
@@ -12,9 +14,15 @@ export const DEFAULT_MODEL_RETRIES = 8;
12
14
  * against a provider that accepts the request and then never streams anything (observed
13
15
  * with throughput-sorted OpenRouter routing), which would otherwise block forever —
14
16
  * `maxRetries` never helps there because the request never errors, it just hangs.
15
- * Overridable per model via `ModelConfig.streamTimeout`.
17
+ * Overridable per model via `ModelConfig.streamTimeout`, or for every model at once via
18
+ * `LlmServiceOptions.streamTimeout` where the application composes its context.
19
+ *
20
+ * Three minutes, not longer: the timer resets on every token, so a legitimately long
21
+ * generation is never at risk — this only bounds SILENCE. The cost of a high value is
22
+ * paid entirely by hung calls, and a hang that takes five minutes to notice is a hang
23
+ * that can stall an agent run for half an hour once retries multiply it.
16
24
  */
17
- export const MODEL_STREAM_TIMEOUT_MS = 5 * 60 * 1000;
25
+ export const MODEL_STREAM_TIMEOUT_MS = 3 * 60 * 1000;
18
26
  /**
19
27
  * Number of failed attempts after which the retry escalator switches from a role's
20
28
  * cheap primary model to its configured `fallback` (stronger) model. With
@@ -29,8 +37,33 @@ export const FALLBACK_AFTER_ATTEMPTS = 3;
29
37
  * issue a 400 "max_tokens exceeds the model's per-request limit".
30
38
  */
31
39
  export const DEFAULT_MAX_OUTPUT_CAP = 3 * 64000;
32
- /** Provider hard limit on prompt-cache breakpoints (Anthropic). */
40
+ /** Provider hard limit on prompt-cache breakpoints per REQUEST (Anthropic). */
33
41
  export const MAX_CACHE_BREAKPOINTS = 4;
42
+ /**
43
+ * Share of {@link MAX_CACHE_BREAKPOINTS} the composed system prompt may spend.
44
+ *
45
+ * Two is all it can use: the only STABLE boundaries are the end of role+skills and the end
46
+ * of the packages block. The trailing context block is volatile and deliberately never
47
+ * marked, so the remaining two breakpoints always stay available to the message prefix.
48
+ */
49
+ export const MAX_SYSTEM_BREAKPOINTS = 2;
50
+ /** Cache lifetime used when neither the call nor the service asks for another. */
51
+ export const DEFAULT_CACHE_TTL = '5m';
52
+ /**
53
+ * Smallest prefix worth marking as cacheable, in tokens. Providers silently refuse to
54
+ * create an entry below their own threshold (1024 tokens on most Claude models, 512 on
55
+ * the newest, 4096 on a few older ones — it is NOT monotonic across generations), so a
56
+ * marker on a short prefix costs nothing but wastes a breakpoint and produces a
57
+ * "caching enabled" log for an entry that was never written. Override per model with
58
+ * `ModelConfig.cacheMinTokens`.
59
+ */
60
+ export const MIN_CACHEABLE_TOKENS = 1024;
61
+ /**
62
+ * Characters per token used to size a prefix against {@link MIN_CACHEABLE_TOKENS}. A rough
63
+ * average for English prose and TypeScript; only ever used to decide whether marking is
64
+ * worth a breakpoint, never for billing or budgeting.
65
+ */
66
+ export const CHARS_PER_TOKEN = 4;
34
67
  /**
35
68
  * Appended to the prompt of `invoke`/`request` when no message already mentions JSON.
36
69
  * Some providers refuse or ignore JSON modes unless the word appears in the prompt;
@@ -71,6 +104,6 @@ export const EFFORT_TABLE = {
71
104
  * without excluding it every `derive`/`escalate`/`withPurpose` would nest another copy.
72
105
  */
73
106
  export const COLLABORATOR_KEYS = [
74
- 'state', 'models', 'model', 'temperatureFactory', 'outputErrors',
107
+ 'state', 'models', 'model', 'temperatureFactory', 'outputErrors', 'files', 'prompts',
75
108
  ];
76
109
  //# sourceMappingURL=consts.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"consts.js","sourceRoot":"","sources":["../src/consts.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,eAAe,EAAE,MAAM,sBAAsB,CAAA;AAGtD,mFAAmF;AACnF,MAAM,CAAC,MAAM,WAAW,GAAG,sBAAsB,CAAA;AAEjD,8DAA8D;AAC9D,MAAM,CAAC,MAAM,iBAAiB,GAAG,gCAAgC,CAAA;AAEjE,6EAA6E;AAC7E,MAAM,CAAC,MAAM,qBAAqB,GAAG,CAAC,CAAA;AAEtC;;;;;;;;GAQG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAA;AAEpD;;;;;GAKG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,CAAC,CAAA;AAExC;;;;;GAKG;AACH,MAAM,CAAC,MAAM,sBAAsB,GAAG,CAAC,GAAG,KAAK,CAAA;AAE/C,mEAAmE;AACnE,MAAM,CAAC,MAAM,qBAAqB,GAAG,CAAC,CAAA;AAEtC;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,6DAA6D;MACzF,gHAAgH;MAChH,iGAAiG;MACjG,2EAA2E,CAAA;AAE/E;;;;GAIG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,WAAW,CAAA;AAE7C,uFAAuF;AACvF,MAAM,CAAC,MAAM,iBAAiB,GAAG,SAAS,CAAA;AAE1C,8DAA8D;AAC9D,MAAM,CAAC,MAAM,cAAc,GAAG,eAAe,CAAC,QAAQ,CAAA;AAEtD;;;;GAIG;AACH,MAAM,CAAC,MAAM,YAAY,GAA8C;IACrE,CAAC,eAAe,CAAC,OAAO,CAAC,EAAE,EAAE,YAAY,EAAE,KAAK,EAAE;IAClD,CAAC,eAAe,CAAC,QAAQ,CAAC,EAAE,EAAE;IAC9B,CAAC,eAAe,CAAC,IAAI,CAAC,EAAE,EAAE,SAAS,EAAE,KAAK,EAAE,YAAY,EAAE,KAAK,EAAE;IACjE,CAAC,eAAe,CAAC,GAAG,CAAC,EAAE,EAAE,SAAS,EAAE,KAAK,EAAE,YAAY,EAAE,KAAK,EAAE;CACjE,CAAA;AAED;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAa;IACzC,OAAO,EAAE,QAAQ,EAAE,OAAO,EAAE,oBAAoB,EAAE,cAAc;CACjE,CAAA"}
1
+ {"version":3,"file":"consts.js","sourceRoot":"","sources":["../src/consts.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,eAAe,EAAE,MAAM,sBAAsB,CAAA;AAGtD,mFAAmF;AACnF,MAAM,CAAC,MAAM,WAAW,GAAG,sBAAsB,CAAA;AAEjD,8DAA8D;AAC9D,MAAM,CAAC,MAAM,iBAAiB,GAAG,gCAAgC,CAAA;AAEjE,0FAA0F;AAC1F,MAAM,CAAC,MAAM,cAAc,GAAG,6BAA6B,CAAA;AAE3D,6EAA6E;AAC7E,MAAM,CAAC,MAAM,qBAAqB,GAAG,CAAC,CAAA;AAEtC;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAA;AAEpD;;;;;GAKG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,CAAC,CAAA;AAExC;;;;;GAKG;AACH,MAAM,CAAC,MAAM,sBAAsB,GAAG,CAAC,GAAG,KAAK,CAAA;AAE/C,+EAA+E;AAC/E,MAAM,CAAC,MAAM,qBAAqB,GAAG,CAAC,CAAA;AAEtC;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,sBAAsB,GAAG,CAAC,CAAA;AAEvC,kFAAkF;AAClF,MAAM,CAAC,MAAM,iBAAiB,GAAa,IAAI,CAAA;AAE/C;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,oBAAoB,GAAG,IAAI,CAAA;AAExC;;;;GAIG;AACH,MAAM,CAAC,MAAM,eAAe,GAAG,CAAC,CAAA;AAEhC;;;;;GAKG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,6DAA6D;MACzF,gHAAgH;MAChH,iGAAiG;MACjG,2EAA2E,CAAA;AAE/E;;;;GAIG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,WAAW,CAAA;AAE7C,uFAAuF;AACvF,MAAM,CAAC,MAAM,iBAAiB,GAAG,SAAS,CAAA;AAE1C,8DAA8D;AAC9D,MAAM,CAAC,MAAM,cAAc,GAAG,eAAe,CAAC,QAAQ,CAAA;AAEtD;;;;GAIG;AACH,MAAM,CAAC,MAAM,YAAY,GAA8C;IACrE,CAAC,eAAe,CAAC,OAAO,CAAC,EAAE,EAAE,YAAY,EAAE,KAAK,EAAE;IAClD,CAAC,eAAe,CAAC,QAAQ,CAAC,EAAE,EAAE;IAC9B,CAAC,eAAe,CAAC,IAAI,CAAC,EAAE,EAAE,SAAS,EAAE,KAAK,EAAE,YAAY,EAAE,KAAK,EAAE;IACjE,CAAC,eAAe,CAAC,GAAG,CAAC,EAAE,EAAE,SAAS,EAAE,KAAK,EAAE,YAAY,EAAE,KAAK,EAAE;CACjE,CAAA;AAED;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAa;IACzC,OAAO,EAAE,QAAQ,EAAE,OAAO,EAAE,oBAAoB,EAAE,cAAc,EAAE,OAAO,EAAE,SAAS;CACrF,CAAA"}
@@ -1 +1 @@
1
- {"version":3,"file":"service.d.ts","sourceRoot":"","sources":["../../src/execution/service.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,MAAM,mBAAmB,CAAA;AAKlE,OAAO,KAAK,EACkB,gBAAgB,EAAE,uBAAuB,EAAE,cAAc,EACrD,oBAAoB,EACrD,MAAM,YAAY,CAAA;AAKnB;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,mBAAmB,GAAI,CAAC,SAAS,cAAc,GAAG,cAAc,EAC3E,SAAS,uBAAuB,EAChC,MAAM,MAAM,gBAAgB,CAAC,CAAC,CAAC,KAC9B,gBAAgB,CAAC,CAAC,CA6HpB,CAAA;AAED,eAAO,MAAM,oBAAoB,GAAI,CAAC,SAAS,cAAc,GAAG,cAAc,EAC5E,QAAO,MAA0B,EACjC,UAAS,uBAA4B,KACpC,gBAAgB,CAAC,CAAC,CAMpB,CAAA;AAED,eAAO,MAAM,sBAAsB,GACjC,CAAC,SAAS,WAAW,EAAE,CAAC,SAAS,YAAY,CAAC,CAAC,CAAC,EAAE,CAAC,SAAS,cAAc,GAAG,cAAc,EAE3F,KAAK,CAAC,EACN,QAAO,MAA0B,EACjC,UAAS,uBAA4B,KACpC,CAAC,GAAG,oBAAoB,CAAC,CAAC,CAQ5B,CAAA"}
1
+ {"version":3,"file":"service.d.ts","sourceRoot":"","sources":["../../src/execution/service.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,MAAM,mBAAmB,CAAA;AAKlE,OAAO,KAAK,EACkB,gBAAgB,EAAE,uBAAuB,EAAE,cAAc,EACrD,oBAAoB,EACrD,MAAM,YAAY,CAAA;AAMnB;;;;;;;;;;;;;;GAcG;AACH,eAAO,MAAM,mBAAmB,GAAI,CAAC,SAAS,cAAc,GAAG,cAAc,EAC3E,SAAS,uBAAuB,EAChC,MAAM,MAAM,gBAAgB,CAAC,CAAC,CAAC,KAC9B,gBAAgB,CAAC,CAAC,CAkIpB,CAAA;AAED,eAAO,MAAM,oBAAoB,GAAI,CAAC,SAAS,cAAc,GAAG,cAAc,EAC5E,QAAO,MAA0B,EACjC,UAAS,uBAA4B,KACpC,gBAAgB,CAAC,CAAC,CAMpB,CAAA;AAED,eAAO,MAAM,sBAAsB,GACjC,CAAC,SAAS,WAAW,EAAE,CAAC,SAAS,YAAY,CAAC,CAAC,CAAC,EAAE,CAAC,SAAS,cAAc,GAAG,cAAc,EAE3F,KAAK,CAAC,EACN,QAAO,MAA0B,EACjC,UAAS,uBAA4B,KACpC,CAAC,GAAG,oBAAoB,CAAC,CAAC,CAQ5B,CAAA"}