@owlmeans/llm 0.1.16-rc.0 → 0.1.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -178,10 +178,9 @@ root `.env`. With none set they self-skip with a printed reason — never a fail
|
|
|
178
178
|
<!-- owlmeans:agent-guidance:start -->
|
|
179
179
|
## Agent guidance
|
|
180
180
|
|
|
181
|
-
This package ships embedded
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
(`.claude/skills/` and `.github/instructions/`):
|
|
181
|
+
This package ships embedded agent skills under `agent-meta/`. After installing your
|
|
182
|
+
`@owlmeans/*` packages, run the OwlMeans agent-skills installer to place them into
|
|
183
|
+
your project's skill store (`.agents/skills/`):
|
|
185
184
|
|
|
186
185
|
```sh
|
|
187
186
|
npx @owlmeans/agent-skills
|
package/agent-meta/manifest.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
|
-
"schemaVersion":
|
|
2
|
+
"schemaVersion": 2,
|
|
3
3
|
"package": "@owlmeans/llm",
|
|
4
|
-
"version": "0.1.16
|
|
5
|
-
"generatedAt": "2026-08-
|
|
4
|
+
"version": "0.1.16",
|
|
5
|
+
"generatedAt": "2026-08-14T10:14:48.851Z",
|
|
6
6
|
"canonicalRepo": "https://github.com/owlmeans/common",
|
|
7
7
|
"entries": [
|
|
8
8
|
{
|
|
@@ -10,28 +10,14 @@
|
|
|
10
10
|
"name": "llm",
|
|
11
11
|
"category": "package-specific",
|
|
12
12
|
"file": "skills/llm/SKILL.md",
|
|
13
|
-
"canonicalPath": ".
|
|
13
|
+
"canonicalPath": ".agents/skills/llm/SKILL.md"
|
|
14
14
|
},
|
|
15
15
|
{
|
|
16
16
|
"kind": "skill",
|
|
17
17
|
"name": "llm-prompt-caching",
|
|
18
18
|
"category": "multi-package",
|
|
19
19
|
"file": "skills/llm-prompt-caching/SKILL.md",
|
|
20
|
-
"canonicalPath": ".
|
|
21
|
-
},
|
|
22
|
-
{
|
|
23
|
-
"kind": "instruction",
|
|
24
|
-
"name": "llm",
|
|
25
|
-
"category": "package-specific",
|
|
26
|
-
"file": "instructions/llm.instructions.md",
|
|
27
|
-
"canonicalPath": ".github/instructions/llm.instructions.md"
|
|
28
|
-
},
|
|
29
|
-
{
|
|
30
|
-
"kind": "instruction",
|
|
31
|
-
"name": "llm-prompt-caching",
|
|
32
|
-
"category": "multi-package",
|
|
33
|
-
"file": "instructions/llm-prompt-caching.instructions.md",
|
|
34
|
-
"canonicalPath": ".github/instructions/llm-prompt-caching.instructions.md"
|
|
20
|
+
"canonicalPath": ".agents/skills/llm-prompt-caching/SKILL.md"
|
|
35
21
|
}
|
|
36
22
|
]
|
|
37
23
|
}
|
|
@@ -8,7 +8,7 @@ user-invocable: false
|
|
|
8
8
|
# @owlmeans/llm
|
|
9
9
|
|
|
10
10
|
**Layer:** Core
|
|
11
|
-
**Install:** `"@owlmeans/llm": "^0.1.16
|
|
11
|
+
**Install:** `"@owlmeans/llm": "^0.1.16"` in `dependencies` (plus the `@langchain/*` peers)
|
|
12
12
|
|
|
13
13
|
The inference runtime. Everything provider-specific is a **plugin**; the model itself only
|
|
14
14
|
owns the provider-independent parts (streaming discipline, retries, validation,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@owlmeans/llm",
|
|
3
|
-
"version": "0.1.16
|
|
3
|
+
"version": "0.1.16",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"scripts": {
|
|
@@ -47,7 +47,7 @@
|
|
|
47
47
|
"@langchain/core": "^1.1.39",
|
|
48
48
|
"@langchain/openai": "^1.4.4",
|
|
49
49
|
"@owlmeans/dep-config": "workspace:*",
|
|
50
|
-
"@owlmeans/test": "^0.1.16
|
|
50
|
+
"@owlmeans/test": "^0.1.16",
|
|
51
51
|
"@types/bun": "^1.3.14",
|
|
52
52
|
"@types/node": "^26.1.0",
|
|
53
53
|
"nodemon": "^3.1.14",
|
|
@@ -55,10 +55,10 @@
|
|
|
55
55
|
},
|
|
56
56
|
"dependencies": {
|
|
57
57
|
"@anthropic-ai/sdk": "^0.78.0",
|
|
58
|
-
"@owlmeans/basic-ids": "^0.1.16
|
|
59
|
-
"@owlmeans/context": "^0.1.16
|
|
60
|
-
"@owlmeans/error": "^0.1.16
|
|
61
|
-
"@owlmeans/llm-common": "^0.1.16
|
|
58
|
+
"@owlmeans/basic-ids": "^0.1.16",
|
|
59
|
+
"@owlmeans/context": "^0.1.16",
|
|
60
|
+
"@owlmeans/error": "^0.1.16",
|
|
61
|
+
"@owlmeans/llm-common": "^0.1.16",
|
|
62
62
|
"ajv": "^8.17.1"
|
|
63
63
|
},
|
|
64
64
|
"publishConfig": {
|
|
@@ -1,100 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
description: "How the OwlMeans LLM layer composes a system prompt from a role and skills, and the prompt-cache rules that layout exists to satisfy — block order, breakpoint budget, determinism invariants, and the provider facts behind them. Consult before changing prompt composition, skills, a prompt plugin, or anything a request sends before its first per-call byte."
|
|
3
|
-
applyTo: "**/*.ts, **/*.tsx"
|
|
4
|
-
---
|
|
5
|
-
<!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
|
|
6
|
-
|
|
7
|
-
# Prompt composition and caching
|
|
8
|
-
|
|
9
|
-
**Layer:** Core · **Packages:** `@owlmeans/llm` (`./prompt`), `@owlmeans/llm-common`, `@owlmeans/agent-skills` (`./llm`)
|
|
10
|
-
|
|
11
|
-
A prompt cache is an **exact prefix match**. Everything below follows from that: the bytes
|
|
12
|
-
every call shares must come first, physically, and must be identical down to the
|
|
13
|
-
whitespace.
|
|
14
|
-
|
|
15
|
-
## The block layout
|
|
16
|
-
|
|
17
|
-
`PromptService.compose()` renders four ordered blocks into one system message:
|
|
18
|
-
|
|
19
|
-
| # | Block | Contributed by | Changes | Breakpoint |
|
|
20
|
-
|---|-------|----------------|---------|-----------|
|
|
21
|
-
| 0 | `Role` | `rolePlugin` ← `PromptPolicy.role` | per role | — |
|
|
22
|
-
| 1 | `Skills` | `skillsPlugin` ← registry + `inline` | per helper | ✅ closes role+skills |
|
|
23
|
-
| 2 | `Packages` | app plugins | per request | ✅ its own |
|
|
24
|
-
| 3 | `Context` | `contextPlugin` ← `context`, `callSkills` | per call | **never** (unless sole block) |
|
|
25
|
-
|
|
26
|
-
A caller's own leading `SystemMessage` is detached by `makeLlmModel` and re-emitted as
|
|
27
|
-
`Context`, so a helper that has not adopted `prompt: { role, skills }` keeps working.
|
|
28
|
-
|
|
29
|
-
## Rules
|
|
30
|
-
|
|
31
|
-
- **Order is the cache key.** `PROMPT_BLOCK_ORDER` is explicit; skills sort by
|
|
32
|
-
`(order, alias)` with a code-unit comparison (never `localeCompare`); detected packages
|
|
33
|
-
sort alphabetically.
|
|
34
|
-
- **Skill bodies are pure constants** — no timestamps, paths, or interpolated request data.
|
|
35
|
-
- **A volatile block is never marked.** The trailing `Context` block changes every
|
|
36
|
-
call: a breakpoint there would pay a cache WRITE on every request and never read
|
|
37
|
-
one back. It is marked ONLY when it is the entire prompt (a caller that has not
|
|
38
|
-
adopted role/skills), because there it IS the stable part. Its parts are merged
|
|
39
|
-
into one chunk for the same reason — nothing downstream needs them separable.
|
|
40
|
-
- **Volatile content goes last.** Per-call skills use `LlmCallOptions.skills` (→ `Context`),
|
|
41
|
-
never the execution's `skills` (→ the cached `Skills` block).
|
|
42
|
-
- **Budget: 4 breakpoints per request.** The system prompt spends at most 3.
|
|
43
|
-
- **The last message is never cached** — it is the per-call payload.
|
|
44
|
-
- **Markers are placed in-place, on the caller's objects.** A caller that carries its
|
|
45
|
-
message array across calls (a coder's growing conversation, a fix loop) hands them back
|
|
46
|
-
still marked, and they accumulate. `prepare()` therefore calls `stripCacheMarkers()`
|
|
47
|
-
first, so the per-request count depends on THIS call alone. Anything that places a
|
|
48
|
-
marker outside that pipeline must do the same.
|
|
49
|
-
- **A prefix below `MIN_CACHEABLE_TOKENS` is left unmarked** (`ModelConfig.cacheMinTokens`
|
|
50
|
-
overrides per alias).
|
|
51
|
-
- **A prompt plugin MUST be deterministic**, and registering the same alias twice replaces
|
|
52
|
-
rather than appends.
|
|
53
|
-
|
|
54
|
-
## Verifying it works
|
|
55
|
-
> **The smoke test cannot catch a caching bug.** It makes no model calls. A breakpoint-budget
|
|
56
|
-
> or marker-accumulation fault only appears under a real multi-call agent run — and it
|
|
57
|
-
> surfaces as a `400`, which the retry loop can bury for minutes. Check the agent's own pod
|
|
58
|
-
> logs for `Prompt cache [` lines and a non-zero `read`.
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
`readCacheUsage(message)` from `@owlmeans/llm/helpers` reads
|
|
62
|
-
`usage_metadata.input_token_details`. If `read` stays 0 across repeated calls that share a
|
|
63
|
-
prefix, something is invalidating it — diff `PromptResult.blocks` between two calls.
|
|
64
|
-
|
|
65
|
-
## Package skills in a prompt (`@owlmeans/agent-skills/llm`)
|
|
66
|
-
|
|
67
|
-
`owlmeansPackagesPlugin(options)` notices which `@owlmeans/*` packages a request mentions
|
|
68
|
-
and loads their published skills into the `Packages` block. Everything a package documents
|
|
69
|
-
is loaded — there is no relevance filtering, by design.
|
|
70
|
-
|
|
71
|
-
Resolution order per package: the host's `LlmFileProvider` (the only path that sees a
|
|
72
|
-
sandbox or remote workspace) → an installed copy under `node_modules` → the canonical
|
|
73
|
-
repository over HTTPS. Every failure is a miss, never a throw. Results, including misses,
|
|
74
|
-
are cached per plugin instance.
|
|
75
|
-
|
|
76
|
-
```typescript
|
|
77
|
-
ctx.prompts().use(owlmeansPackagesPlugin({
|
|
78
|
-
files: () => ctx.files(), // tried first
|
|
79
|
-
exclude: ['@owlmeans/llm'], // already covered by the static Skills block
|
|
80
|
-
fetch: false, // air-gapped: skip the repository fallback
|
|
81
|
-
}))
|
|
82
|
-
```
|
|
83
|
-
|
|
84
|
-
The manifest deliberately carries no git ref — version-matching comes from shipping the
|
|
85
|
-
copy inside the tarball, so for a package that is NOT installed the ref is a plugin option
|
|
86
|
-
(`ref`, default `main`).
|
|
87
|
-
|
|
88
|
-
## Provider facts these rules encode
|
|
89
|
-
|
|
90
|
-
**Anthropic** ([docs](https://platform.claude.com/docs/en/build-with-claude/prompt-caching.md)):
|
|
91
|
-
render order `tools` → `system` → `messages`; max 4 breakpoints; minimum cacheable prefix is
|
|
92
|
-
model-dependent and **not monotonic** (512 newest / 1024 most / 4096 on Opus 4.6-4.5 and
|
|
93
|
-
Haiku 4.5); writes cost ~1.25× at 5m TTL and ~2× at 1h; a `system` change invalidates
|
|
94
|
-
system + messages but not `tools`.
|
|
95
|
-
|
|
96
|
-
**OpenAI** ([docs](https://developers.openai.com/api/docs/guides/prompt-caching)): automatic
|
|
97
|
-
from 1024 tokens; routing hashes roughly the first 256 tokens and mixes in
|
|
98
|
-
`prompt_cache_key`. The `openai` plugin sets it from the config alias; the `compatible`
|
|
99
|
-
plugin does not, because aggregators with `provider.require_parameters` can drop providers
|
|
100
|
-
over an unknown field.
|
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
description: "How to use @owlmeans/llm — the LLM inference runtime: four-method model, provider plugins, model factory service, the policy-driven execution abstraction, and the prompt/skill composition service."
|
|
3
|
-
applyTo: "**/*.ts, **/*.tsx"
|
|
4
|
-
---
|
|
5
|
-
<!-- AUTO-GENERATED — do not edit. Regenerate via sync-agent-meta. -->
|
|
6
|
-
|
|
7
|
-
# @owlmeans/llm
|
|
8
|
-
|
|
9
|
-
**Layer:** Core
|
|
10
|
-
**Install:** `"@owlmeans/llm": "^0.1.16-rc.0"` in `dependencies` (plus the `@langchain/*` peers)
|
|
11
|
-
|
|
12
|
-
The inference runtime. Everything provider-specific is a **plugin**; the model owns only the
|
|
13
|
-
provider-independent parts. Serializable contracts live in `@owlmeans/llm-common`.
|
|
14
|
-
|
|
15
|
-
## Key Exports
|
|
16
|
-
|
|
17
|
-
| Export | Description |
|
|
18
|
-
|--------|-------------|
|
|
19
|
-
| `makeLlmModel(options, spectator)` | `ask` / `talk` / `invoke` / `request`. |
|
|
20
|
-
| `makeLlmService` · `appendLlmService` · `llmServiceApi` | Model factory/registry; `llmServiceApi` composes into your own service. |
|
|
21
|
-
| `makeExecutionService` · `appendExecutionService` · `executionServiceApi` | Frozen 3-level executions, policy resolution, snapshot/restore/checkpoint. |
|
|
22
|
-
| `makePromptService` · `appendPromptService` · `promptServiceApi` | Skill registry + composition plugin chain (`@owlmeans/llm/prompt`). |
|
|
23
|
-
| `renderSkill`, `sortSkills`, `compareAlias`, `readCacheUsage` | Deterministic rendering and prompt-cache accounting. |
|
|
24
|
-
| `plugins`, `registerLlmPlugin`, `resolvePlugin`, `pluginOf`, `pluginFor` | Provider-plugin registry (`@owlmeans/llm/plugins`). |
|
|
25
|
-
| `anthropicPlugin`, `openAiPlugin`, `compatiblePlugin`, `openAiFamily` | Built-ins; spread `openAiFamily` into a new OpenAI-compatible plugin. |
|
|
26
|
-
| `withRetry`, `registerFatalError`, `spectate`, `normalizeInput`, `parseJsonContent`, `coerceToSchema` | Helpers (`@owlmeans/llm/helpers`). |
|
|
27
|
-
| `LlmModelError` (retryable), `LlmMissconfiguredError`, `LlmPluginError`, `LlmRetryExceededError` | `ResilientError` family. |
|
|
28
|
-
|
|
29
|
-
## Rules
|
|
30
|
-
|
|
31
|
-
- `src/helpers/` is what consumers may use; `src/utils/` is library-private and never
|
|
32
|
-
exported. Place a new function on the right side rather than exporting a util for a test.
|
|
33
|
-
- Never branch on the provider (`instanceof ChatAnthropic`, `provider === …`) in the model or
|
|
34
|
-
the service — put it on the `LlmPlugin` (`build` / `owns` / `family` / `refine` /
|
|
35
|
-
`structuredMode` / `toolChoice` / `responseFormat` / `patchSystem` / `patchCache` / `isFatal`).
|
|
36
|
-
- Plugin registration order is load-bearing: `pluginFor` returns the first `owns` match, and
|
|
37
|
-
`compatible` precedes `openai` so an unlabelled `ChatOpenAI` gets the conservative
|
|
38
|
-
tool-calling behaviour.
|
|
39
|
-
- Never `instanceof` an error from a provider SDK: `@langchain/*` bundle their own
|
|
40
|
-
nested copies, so the check silently fails and a fatal `400` gets retried eight
|
|
41
|
-
times. Match `status === 400` instead.
|
|
42
|
-
- `@langchain/*` are **peer** dependencies — model instances cross the package boundary and
|
|
43
|
-
two installed copies are nominally distinct types. Pin one copy in the consumer.
|
|
44
|
-
- Extend the execution generically: your own `ExecutionShape`, your collaborator fields in
|
|
45
|
-
`collaboratorKeys`. Do not narrow inherited method signatures (contravariance error).
|
|
46
|
-
- Never hand-build a persona as a `SystemMessage` in a helper. Declare it as
|
|
47
|
-
`PromptPolicy.role` plus registered skills and let the prompt service compose it —
|
|
48
|
-
otherwise the knowledge duplicates and the cacheable prefix stops being stable. Skills
|
|
49
|
-
accumulate project → task → helper; the deepest declared role wins (`mergePrompt`).
|
|
50
|
-
Block order, breakpoint budget and the provider facts: `llm-prompt-caching`.
|
|
51
|
-
|
|
52
|
-
## Usage
|
|
53
|
-
|
|
54
|
-
```typescript
|
|
55
|
-
import { makeLlmModel, makeLlmService } from '@owlmeans/llm'
|
|
56
|
-
|
|
57
|
-
const llm = makeLlmService({ models: () => configs })
|
|
58
|
-
const model = makeLlmModel(
|
|
59
|
-
{ model: llm.getModel('analyst'), purpose: { type: 'analysis' } }, spectator
|
|
60
|
-
)
|
|
61
|
-
const spec = await model.invoke('Describe the app', SpecSchema, { action: 'spec' })
|
|
62
|
-
```
|
|
63
|
-
|
|
64
|
-
Execution resolution precedence: **roleOverride → modelOverride → effort tier →
|
|
65
|
-
`LlmService.getModel`**.
|
|
66
|
-
|
|
67
|
-
## Already handled — do not reimplement
|
|
68
|
-
|
|
69
|
-
Idle stream deadline · duplicate-final-chunk dedup · output-budget escalation ·
|
|
70
|
-
reasoning-cap shrink · same-family fallback model · schema coercion · JSON salvage from
|
|
71
|
-
prose · `NullCapture` diagnostics · fatal-error short-circuit.
|
|
72
|
-
|
|
73
|
-
## Depends On
|
|
74
|
-
|
|
75
|
-
- `@owlmeans/llm-common`, `@owlmeans/context`, `@owlmeans/error`, `@owlmeans/basic-ids`, `ajv`
|
|
76
|
-
- peer `@langchain/core`, `@langchain/openai`, `@langchain/anthropic`
|