tonst 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tonst/__init__.py ADDED
@@ -0,0 +1,81 @@
1
+ from .client import TonstClient, OptimizationReport, StructuredRedactionResult
2
+ from .cache_structuring import (
3
+ PromptParts,
4
+ structure_for_caching,
5
+ build_anthropic_cache_request,
6
+ check_cache_eligibility,
7
+ CacheEligibility,
8
+ parse_anthropic_usage,
9
+ CacheUsageReport,
10
+ )
11
+ from .compactor import (
12
+ HistoryCompactor,
13
+ compact_history,
14
+ CompactionResult,
15
+ compact_history_rolling,
16
+ RollingSummary,
17
+ RollingCompactionResult,
18
+ FoldJob,
19
+ run_fold_job,
20
+ estimate_fold_payback,
21
+ )
22
+ from .trim import flatten_messages
23
+ from .tool_optimizer import (
24
+ select_tools,
25
+ ToolSelection,
26
+ ToolSession,
27
+ build_anthropic_deferred_tools,
28
+ estimate_tool_tokens,
29
+ DEFERRED_TOOLS_SYSTEM_HINT,
30
+ )
31
+ from .rag import optimize_chunks, ChunkSelection
32
+ from .token_count import AnthropicTokenCounter, GeminiTokenCounter
33
+ from .summarizers import AnthropicSummarizer, GeminiSummarizer
34
+ from .savings_log import SavingsLog, summarize as summarize_savings, SavingsSummary
35
+ from .placeholders import PlaceholderFactory
36
+ from .redact import StreamRestorer, restore_placeholders
37
+ from . import providers
38
+ from . import adapters
39
+
40
+ __all__ = [
41
+ "PlaceholderFactory",
42
+ "restore_placeholders",
43
+ "StreamRestorer",
44
+ "TonstClient",
45
+ "OptimizationReport",
46
+ "StructuredRedactionResult",
47
+ "PromptParts",
48
+ "structure_for_caching",
49
+ "build_anthropic_cache_request",
50
+ "check_cache_eligibility",
51
+ "CacheEligibility",
52
+ "parse_anthropic_usage",
53
+ "CacheUsageReport",
54
+ "HistoryCompactor",
55
+ "compact_history",
56
+ "CompactionResult",
57
+ "compact_history_rolling",
58
+ "RollingSummary",
59
+ "RollingCompactionResult",
60
+ "FoldJob",
61
+ "run_fold_job",
62
+ "estimate_fold_payback",
63
+ "flatten_messages",
64
+ "select_tools",
65
+ "ToolSelection",
66
+ "ToolSession",
67
+ "build_anthropic_deferred_tools",
68
+ "estimate_tool_tokens",
69
+ "DEFERRED_TOOLS_SYSTEM_HINT",
70
+ "optimize_chunks",
71
+ "ChunkSelection",
72
+ "AnthropicTokenCounter",
73
+ "GeminiTokenCounter",
74
+ "AnthropicSummarizer",
75
+ "GeminiSummarizer",
76
+ "SavingsLog",
77
+ "summarize_savings",
78
+ "SavingsSummary",
79
+ "providers",
80
+ "adapters",
81
+ ]
tonst/__main__.py ADDED
@@ -0,0 +1,43 @@
1
+ """
2
+ Command-line entry point.
3
+
4
+ tonst stats [--log PATH] [--app NAME] [--since YYYY-MM-DD] [--json]
5
+
6
+ Also runnable without installing the console script:
7
+
8
+ python -m tonst stats
9
+ """
10
+
11
+ from __future__ import annotations
12
+ import argparse
13
+ import json
14
+ import sys
15
+
16
+ from .savings_log import summarize, format_summary
17
+
18
+
19
+ def main(argv=None) -> int:
20
+ parser = argparse.ArgumentParser(prog="tonst", description="tonst command-line tools")
21
+ sub = parser.add_subparsers(dest="command")
22
+
23
+ stats = sub.add_parser("stats", help="summarize the local savings log")
24
+ stats.add_argument("--log", default=None, help="log path (default: $TONST_SAVINGS_LOG or ~/.tonst/savings.jsonl)")
25
+ stats.add_argument("--app", default=None, help="only include entries for this app name")
26
+ stats.add_argument("--since", default=None, help="only include entries on/after this ISO date, e.g. 2026-09-01")
27
+ stats.add_argument("--json", action="store_true", help="print machine-readable JSON")
28
+
29
+ args = parser.parse_args(argv)
30
+ if args.command != "stats":
31
+ parser.print_help()
32
+ return 1
33
+
34
+ summary = summarize(args.log, app=args.app, since=args.since)
35
+ if args.json:
36
+ print(json.dumps(summary.to_dict(), indent=2))
37
+ else:
38
+ print(format_summary(summary))
39
+ return 0
40
+
41
+
42
+ if __name__ == "__main__":
43
+ sys.exit(main())
tonst/adapters.py ADDED
@@ -0,0 +1,133 @@
1
+ """
2
+ adapters.py
3
+ -----------
4
+ Small converters between tonst's plain message list and each provider's
5
+ request/response shapes, for TonstClient(messages_fn=...).
6
+
7
+ tonst hands messages_fn the final, redacted conversation as
8
+ [{"role": "system" | "user" | "assistant", "content": str}, ...].
9
+ These helpers turn that into what each API expects, and turn the API's
10
+ usage block back into the {"prompt_tokens", "cached_tokens"} dict tonst
11
+ reads (so the report and cache-aware compaction see real numbers).
12
+
13
+ import anthropic
14
+ from tonst import TonstClient
15
+ from tonst.adapters import to_anthropic, usage_from_anthropic
16
+
17
+ api = anthropic.Anthropic()
18
+
19
+ def call_claude(messages):
20
+ resp = api.messages.create(model="claude-sonnet-4-6", max_tokens=1024,
21
+ **to_anthropic(messages))
22
+ return resp.content[0].text, usage_from_anthropic(resp)
23
+
24
+ client = TonstClient(messages_fn=call_claude)
25
+
26
+ Every helper accepts either a plain dict (raw JSON) or an SDK response
27
+ object, and none of them imports a provider SDK.
28
+ """
29
+
30
+ from __future__ import annotations
31
+ from typing import Optional
32
+
33
+
34
+ def _get(obj, *path, default=None):
35
+ """obj.a.b or obj["a"]["b"], whichever exists."""
36
+ cur = obj
37
+ for key in path:
38
+ if cur is None:
39
+ return default
40
+ if isinstance(cur, dict):
41
+ cur = cur.get(key)
42
+ else:
43
+ cur = getattr(cur, key, None)
44
+ return default if cur is None else cur
45
+
46
+
47
+ def _merge_turns(messages: list, assistant_role: str) -> tuple:
48
+ """(system text, [(role, [texts])]) with same-role neighbours merged and a user turn first."""
49
+ system = "\n\n".join(m["content"] for m in messages if m.get("role") == "system" and m.get("content"))
50
+ turns = []
51
+ for m in messages:
52
+ role = m.get("role")
53
+ if role == "system" or not m.get("content"):
54
+ continue
55
+ role = assistant_role if role == "assistant" else "user"
56
+ if turns and turns[-1][0] == role:
57
+ turns[-1][1].append(m["content"])
58
+ else:
59
+ turns.append((role, [m["content"]]))
60
+ if turns and turns[0][0] != "user":
61
+ turns.insert(0, ("user", ["(Earlier conversation omitted.)"]))
62
+ return system, turns
63
+
64
+
65
+ def to_anthropic(messages: list, cache: bool = True) -> dict:
66
+ """
67
+ Keyword arguments for Anthropic's Messages API: {"system", "messages"}.
68
+ Roles must alternate there, so same-role neighbours are merged (tonst's
69
+ summary message is a user turn and can sit next to one). cache=True puts
70
+ a cache_control breakpoint on the system prompt and a moving one on the
71
+ last block, which is how a multi-turn chat gets cached turn after turn.
72
+ Prompts under the model's minimum cacheable length are simply not cached.
73
+ """
74
+ system, turns = _merge_turns(messages, "assistant")
75
+ out_msgs = [{"role": r, "content": [{"type": "text", "text": t} for t in texts]} for r, texts in turns]
76
+ if cache and out_msgs:
77
+ last = out_msgs[-1]["content"][-1]
78
+ out_msgs[-1]["content"][-1] = {**last, "cache_control": {"type": "ephemeral"}}
79
+ body = {"messages": out_msgs}
80
+ if system:
81
+ block = {"type": "text", "text": system}
82
+ if cache:
83
+ block["cache_control"] = {"type": "ephemeral"}
84
+ body["system"] = [block]
85
+ return body
86
+
87
+
88
+ def to_openai(messages: list) -> list:
89
+ """Chat Completions `messages`. OpenAI caches repeated prefixes automatically; no markers needed."""
90
+ return [{"role": m["role"], "content": m["content"]} for m in messages if m.get("content")]
91
+
92
+
93
+ def to_gemini(messages: list) -> dict:
94
+ """
95
+ generateContent REST body fields: {"systemInstruction", "contents"}
96
+ (role "model" for the assistant). Gemini 2.5+ caches repeated prefixes
97
+ implicitly; it's best-effort and needs a few thousand tokens.
98
+ """
99
+ system, turns = _merge_turns(messages, "model")
100
+ body = {"contents": [{"role": r, "parts": [{"text": t} for t in texts]} for r, texts in turns]}
101
+ if system:
102
+ body["systemInstruction"] = {"parts": [{"text": system}]}
103
+ return body
104
+
105
+
106
+ def usage_from_anthropic(resp) -> Optional[dict]:
107
+ u = _get(resp, "usage")
108
+ if u is None:
109
+ return None
110
+ read = int(_get(u, "cache_read_input_tokens", default=0))
111
+ total = int(_get(u, "input_tokens", default=0)) + int(_get(u, "cache_creation_input_tokens", default=0)) + read
112
+ return {"prompt_tokens": total, "cached_tokens": read}
113
+
114
+
115
+ def usage_from_openai(resp) -> Optional[dict]:
116
+ u = _get(resp, "usage")
117
+ if u is None:
118
+ return None
119
+ if _get(u, "prompt_tokens") is not None: # Chat Completions
120
+ return {"prompt_tokens": int(_get(u, "prompt_tokens")),
121
+ "cached_tokens": int(_get(u, "prompt_tokens_details", "cached_tokens", default=0))}
122
+ return {"prompt_tokens": int(_get(u, "input_tokens", default=0)), # Responses API
123
+ "cached_tokens": int(_get(u, "input_tokens_details", "cached_tokens", default=0))}
124
+
125
+
126
+ def usage_from_gemini(resp) -> Optional[dict]:
127
+ u = _get(resp, "usageMetadata") or _get(resp, "usage_metadata")
128
+ if u is None:
129
+ return None
130
+ prompt = _get(u, "promptTokenCount") if _get(u, "promptTokenCount") is not None else _get(u, "prompt_token_count", default=0)
131
+ cached = _get(u, "cachedContentTokenCount") if _get(u, "cachedContentTokenCount") is not None \
132
+ else _get(u, "cached_content_token_count", default=0)
133
+ return {"prompt_tokens": int(prompt), "cached_tokens": int(cached or 0)}
@@ -0,0 +1,405 @@
1
+ """
2
+ cache_structuring.py
3
+ ---------------------
4
+ Structures a request so that provider-side prompt caching actually pays
5
+ off. Important to be precise about what this is NOT: prompt caching does
6
+ not skip the LLM call. Anthropic, OpenAI, and Gemini all cache the
7
+ *processed representation* of repeated prefix tokens so the model
8
+ doesn't have to reprocess them, but the model still runs and generates a
9
+ fresh response on every call -- caching only discounts (and speeds up)
10
+ the repeated portion of the input.
11
+
12
+ It only pays off if the request is shaped correctly:
13
+ - Stable, reused content (system instructions, a reference document,
14
+ tool descriptions written out as text) has to come FIRST.
15
+ - That stable content has to be byte-for-byte IDENTICAL across calls --
16
+ reordering it, changing its whitespace, or interleaving the variable
17
+ question into it all silently break the match. There's no error when
18
+ this happens; you just quietly stop getting the discount.
19
+ - The variable part (the actual question, the newest turn) goes LAST.
20
+ - The stable prefix also has to clear a PER-MODEL MINIMUM LENGTH --
21
+ see CACHE_MINIMUM_TOKENS below. Below it, Anthropic silently skips
22
+ caching entirely: no error, cache_creation_input_tokens and
23
+ cache_read_input_tokens both stay 0. check_cache_eligibility() and
24
+ the warning in build_anthropic_cache_request() exist specifically so
25
+ this doesn't fail silently on you too.
26
+
27
+ Output shapes provided:
28
+ - structure_for_caching() returns a flat, correctly-ordered string.
29
+ This is enough for providers with automatic prefix caching (no
30
+ explicit marker required) and works with TonstClient.query(str).
31
+ - build_anthropic_cache_request() returns the actual Anthropic
32
+ Messages API JSON body with an explicit `cache_control` breakpoint,
33
+ which Anthropic requires -- correct ordering alone is not sufficient
34
+ on that provider. Verified against Anthropic's prompt-caching docs
35
+ (platform.claude.com/docs/en/build-with-claude/prompt-caching,
36
+ checked 2026-09-08): cache_control goes on the LAST block whose
37
+ prefix should be cached, as {"type": "ephemeral", "ttl": "5m"|"1h"}.
38
+ - parse_anthropic_usage() reads the `usage` field of a REAL API
39
+ response and reports whether a cache hit actually happened --
40
+ the only ground truth for whether any of this worked.
41
+ """
42
+
43
+ from __future__ import annotations
44
+ import warnings
45
+ from dataclasses import dataclass, field
46
+ from typing import Optional
47
+
48
+ VALID_TTLS = ("5m", "1h")
49
+
50
+ # Anthropic's minimum cacheable-prefix length per model, in tokens, as of
51
+ # the docs checked 2026-09-08. A stable prefix shorter than this is
52
+ # processed normally with NO ERROR and NO CACHING -- the response's
53
+ # usage.cache_creation_input_tokens / usage.cache_read_input_tokens both
54
+ # come back 0. This table exists so tonst can warn about that ahead of
55
+ # time instead of a developer discovering it by silently getting no
56
+ # savings. Keyed on the model string as passed to the API; unknown/future
57
+ # models fall back to DEFAULT_CACHE_MINIMUM_TOKENS (the most common
58
+ # figure across the current lineup).
59
+ CACHE_MINIMUM_TOKENS: dict[str, int] = {
60
+ "claude-haiku-4-5": 4096,
61
+ "claude-haiku-3-5": 2048,
62
+ "claude-opus-4-6": 4096,
63
+ "claude-opus-4-5": 4096,
64
+ "claude-opus-4-7": 2048,
65
+ "claude-opus-4-8": 1024,
66
+ "claude-opus-4-1": 1024,
67
+ "claude-opus-4": 1024,
68
+ "claude-sonnet-5": 1024,
69
+ "claude-sonnet-4-6": 1024,
70
+ "claude-sonnet-4-5": 1024,
71
+ "claude-sonnet-4": 1024,
72
+ "claude-opus-5": 512,
73
+ "claude-fable-5": 512,
74
+ "claude-fable-5-1": 512,
75
+ "claude-mythos-5": 512,
76
+ "claude-mythos-5-1": 512,
77
+ # "Claude Mythos Preview" is documented at 2,048 tokens, but its
78
+ # exact API model-id string wasn't confirmed at the time this table
79
+ # was written -- add it here once known, rather than guess a key
80
+ # that would silently never match.
81
+ }
82
+ DEFAULT_CACHE_MINIMUM_TOKENS = 1024
83
+
84
+ # Anthropic's cache pricing multipliers, relative to that model's own
85
+ # base (uncached) input-token price of 1.0x. Verified against
86
+ # platform.claude.com/docs/en/build-with-claude/prompt-caching, checked
87
+ # 2026-09-09. A cache WRITE costs MORE than a fresh token -- you're
88
+ # paying a premium to populate the cache -- and only a cache READ is
89
+ # discounted. This is what makes "percent of input from cache" (see
90
+ # CacheUsageReport.percent_of_input_from_cache) different from an actual
91
+ # cost saving: caching never changes how many tokens get processed, so
92
+ # a real savings estimate has to weight each token type by its actual
93
+ # price, not just count what fraction came from cache.
94
+ CACHE_READ_MULTIPLIER_DEFAULT = 0.1 # 10% of base input price
95
+ CACHE_READ_MULTIPLIER_LOW_COST = 0.025 # Fable/Mythos family: 2.5% of base price
96
+ CACHE_WRITE_MULTIPLIER_5M = 1.25 # 125% of base input price
97
+ CACHE_WRITE_MULTIPLIER_1H = 2.0 # 200% of base input price
98
+
99
+ # Models with the lower 2.5% cache-read rate instead of the standard 10%.
100
+ _LOW_COST_CACHE_READ_MODELS = {
101
+ "claude-fable-5",
102
+ "claude-fable-5-1",
103
+ "claude-mythos-5",
104
+ "claude-mythos-5-1",
105
+ }
106
+
107
+
108
+ def _cache_read_multiplier(model: str) -> float:
109
+ return (
110
+ CACHE_READ_MULTIPLIER_LOW_COST
111
+ if model in _LOW_COST_CACHE_READ_MODELS
112
+ else CACHE_READ_MULTIPLIER_DEFAULT
113
+ )
114
+
115
+
116
+ @dataclass
117
+ class PromptParts:
118
+ """
119
+ system: instructions that never change between calls in this
120
+ conversation/session (e.g. "You are a support assistant...").
121
+ stable_blocks: large reusable content that's IDENTICAL across many
122
+ calls -- reference documents, few-shot examples, tool
123
+ descriptions written out as text. List order is preserved; put
124
+ the content least likely to ever change first, since everything
125
+ up to and including the cache_control breakpoint must match
126
+ exactly for a hit.
127
+ variable: the part that's different on every call -- the user's
128
+ actual question, or the newest turn. Always placed last.
129
+ """
130
+ system: Optional[str] = None
131
+ stable_blocks: list = field(default_factory=list)
132
+ variable: str = ""
133
+
134
+
135
+ def structure_for_caching(parts: PromptParts) -> str:
136
+ """
137
+ Returns a single string with stable content first, variable last.
138
+ Byte-for-byte identical stable content across calls is what lets
139
+ providers with automatic prefix caching (no explicit marker needed)
140
+ actually get a cache hit -- reordering, whitespace changes, or
141
+ interleaving the question into the stable section all silently
142
+ defeat it. This is the ordering guarantee; it does not, by itself,
143
+ set an explicit cache breakpoint (see build_anthropic_cache_request
144
+ for that).
145
+ """
146
+ sections = []
147
+ if parts.system:
148
+ sections.append(parts.system.strip())
149
+ sections.extend(b.strip() for b in parts.stable_blocks if b and b.strip())
150
+ if parts.variable:
151
+ sections.append(parts.variable.strip())
152
+ return "\n\n".join(sections)
153
+
154
+
155
+ @dataclass
156
+ class CacheEligibility:
157
+ """
158
+ Heuristic estimate of whether `parts`'s stable content is long enough
159
+ for `model` to actually cache it. This uses tonst's chars/4 token
160
+ estimate (see trim.py), NOT the provider's real tokenizer -- treat it
161
+ as a smoke test that catches the obvious "way too short" case, not
162
+ an exact prediction. The only real ground truth is the `usage` field
163
+ of an actual API response (see parse_anthropic_usage()).
164
+ """
165
+ eligible: bool
166
+ stable_tokens_estimate: int
167
+ minimum_required: int
168
+ model: str
169
+
170
+ @property
171
+ def message(self) -> str:
172
+ if self.eligible:
173
+ return (
174
+ f"Stable content is ~{self.stable_tokens_estimate} tokens "
175
+ f"(est.), above {self.model}'s {self.minimum_required}-token "
176
+ "minimum -- likely eligible for caching."
177
+ )
178
+ return (
179
+ f"Stable content is only ~{self.stable_tokens_estimate} tokens "
180
+ f"(est.), below {self.model}'s {self.minimum_required}-token "
181
+ "minimum. This provider will NOT cache it: most providers raise "
182
+ "no error here, they simply process the request without "
183
+ "caching. Add more reusable content to stable_blocks/system, "
184
+ "or don't bother shaping this request for caching at all."
185
+ )
186
+
187
+
188
+ def check_cache_eligibility(
189
+ parts: PromptParts, model: str, token_estimator=None, tools: Optional[list] = None
190
+ ) -> CacheEligibility:
191
+ """
192
+ Estimates whether the stable/cacheable portion of `parts` meets
193
+ Anthropic's minimum prefix length for `model`. Meant to catch an
194
+ obviously-too-short system prompt or reference snippet before
195
+ spending an API call that silently won't cache -- not a substitute
196
+ for checking real usage.cache_creation_input_tokens /
197
+ usage.cache_read_input_tokens from an actual response.
198
+
199
+ `tools`, if given, counts toward the cached prefix too (tool
200
+ definitions come first in Anthropic's prefix order) -- except tools
201
+ marked defer_loading, which the API keeps out of the prefix.
202
+ """
203
+ if token_estimator is None:
204
+ from .trim import estimate_tokens as token_estimator
205
+
206
+ stable_text = "\n\n".join([parts.system or ""] + list(parts.stable_blocks))
207
+ stable_tokens = token_estimator(stable_text) if stable_text.strip() else 0
208
+ loaded = [t for t in (tools or []) if not t.get("defer_loading")]
209
+ if loaded:
210
+ import json
211
+ stable_tokens += token_estimator(json.dumps(loaded, separators=(",", ":"), sort_keys=True))
212
+
213
+ minimum = CACHE_MINIMUM_TOKENS.get(model, DEFAULT_CACHE_MINIMUM_TOKENS)
214
+ return CacheEligibility(
215
+ eligible=stable_tokens >= minimum,
216
+ stable_tokens_estimate=stable_tokens,
217
+ minimum_required=minimum,
218
+ model=model,
219
+ )
220
+
221
+
222
+ def build_anthropic_cache_request(
223
+ parts: PromptParts,
224
+ model: str,
225
+ max_tokens: int = 1000,
226
+ cache_ttl: str = "5m",
227
+ warn_if_ineligible: bool = True,
228
+ tools: Optional[list] = None,
229
+ ) -> dict:
230
+ """
231
+ Builds the JSON body for a direct call to
232
+ https://api.anthropic.com/v1/messages with an explicit cache
233
+ breakpoint placed after the stable content. Anthropic requires this
234
+ `cache_control` marker on the last stable block -- ordering alone
235
+ (unlike OpenAI/Gemini's automatic prefix caching) does not trigger a
236
+ cache hit on this provider.
237
+
238
+ If `warn_if_ineligible` is True (default), this checks stable-content
239
+ length against CACHE_MINIMUM_TOKENS and raises a UserWarning if it's
240
+ likely too short to actually be cached. The request is still built
241
+ and returned either way -- Anthropic itself doesn't error on a
242
+ too-short cache_control block, it just silently skips caching, and
243
+ tonst matches that (fail-soft), just with a visible warning instead
244
+ of silence.
245
+
246
+ If `parts.system` is set, it's cached too (as its own block with its
247
+ own breakpoint) since a large, reused system prompt is one of the
248
+ most common things worth caching.
249
+
250
+ If there are no stable_blocks, this falls back to a plain string
251
+ message body -- there's nothing to cache, so the block-array
252
+ overhead (and the cost of writing a cache entry with no reuse ahead
253
+ of it) isn't worth it.
254
+
255
+ `tools` (optional) is passed through as the request's `tools` array,
256
+ unmodified -- e.g. the output of tool_optimizer.select_tools() or
257
+ build_anthropic_deferred_tools(). Tools come FIRST in Anthropic's
258
+ cache prefix, so the system breakpoint already covers them; when
259
+ there's no system prompt, a breakpoint is put on the last
260
+ non-deferred tool instead so the tool definitions still get cached.
261
+ """
262
+ if cache_ttl not in VALID_TTLS:
263
+ raise ValueError(f"cache_ttl must be one of {VALID_TTLS}, got {cache_ttl!r}")
264
+
265
+ if warn_if_ineligible:
266
+ eligibility = check_cache_eligibility(parts, model, tools=tools)
267
+ if not eligibility.eligible:
268
+ warnings.warn(eligibility.message, UserWarning, stacklevel=2)
269
+
270
+ body: dict = {"model": model, "max_tokens": max_tokens}
271
+
272
+ if tools:
273
+ for t in tools:
274
+ if t.get("defer_loading") and "cache_control" in t:
275
+ raise ValueError(
276
+ f"tool {t.get('name')!r} has both defer_loading and cache_control; "
277
+ "Anthropic rejects that with a 400"
278
+ )
279
+ body["tools"] = [dict(t) for t in tools]
280
+ if not parts.system:
281
+ loaded_idx = [i for i, t in enumerate(body["tools"]) if not t.get("defer_loading")]
282
+ if loaded_idx and not any("cache_control" in t for t in body["tools"]):
283
+ body["tools"][loaded_idx[-1]]["cache_control"] = {"type": "ephemeral", "ttl": cache_ttl}
284
+
285
+ if parts.system:
286
+ body["system"] = [
287
+ {
288
+ "type": "text",
289
+ "text": parts.system.strip(),
290
+ "cache_control": {"type": "ephemeral", "ttl": cache_ttl},
291
+ }
292
+ ]
293
+
294
+ stable = [b.strip() for b in parts.stable_blocks if b and b.strip()]
295
+ variable_text = (parts.variable or "").strip()
296
+
297
+ if stable:
298
+ content_blocks = [{"type": "text", "text": b} for b in stable]
299
+ # The breakpoint goes on the LAST stable block: everything up to
300
+ # and including it becomes the cached prefix, so only one marker
301
+ # is needed even if there are many stable blocks.
302
+ content_blocks[-1]["cache_control"] = {"type": "ephemeral", "ttl": cache_ttl}
303
+ if variable_text:
304
+ content_blocks.append({"type": "text", "text": variable_text})
305
+ body["messages"] = [{"role": "user", "content": content_blocks}]
306
+ else:
307
+ # Nothing stable to cache -- a plain string is simpler and avoids
308
+ # block-array overhead for a one-off call.
309
+ body["messages"] = [{"role": "user", "content": variable_text}]
310
+
311
+ return body
312
+
313
+
314
+ @dataclass
315
+ class CacheUsageReport:
316
+ """
317
+ Parsed from a REAL Anthropic API response's `usage` field -- the
318
+ only ground truth for whether caching actually happened on a given
319
+ call. cache_creation_input_tokens > 0 means this call WROTE a new
320
+ cache entry (expected on the first call with a given stable prefix).
321
+ cache_read_input_tokens > 0 means this call GOT the discount by
322
+ reading a previously-written cache entry.
323
+ """
324
+ input_tokens: int
325
+ output_tokens: int
326
+ cache_creation_input_tokens: int
327
+ cache_read_input_tokens: int
328
+
329
+ @property
330
+ def cache_hit(self) -> bool:
331
+ return self.cache_read_input_tokens > 0
332
+
333
+ @property
334
+ def cache_write(self) -> bool:
335
+ return self.cache_creation_input_tokens > 0
336
+
337
+ @property
338
+ def percent_of_input_from_cache(self) -> float:
339
+ total_input = (
340
+ self.input_tokens
341
+ + self.cache_creation_input_tokens
342
+ + self.cache_read_input_tokens
343
+ )
344
+ if total_input == 0:
345
+ return 0.0
346
+ return round(100 * self.cache_read_input_tokens / total_input, 1)
347
+
348
+ def estimated_cost_savings_percent(self, model: str, cache_ttl: str = "5m") -> float:
349
+ """
350
+ Estimates the % COST difference caching made on this call,
351
+ against the baseline of the exact same token count being sent
352
+ with NO caching at all (everything billed at the standard 1x
353
+ input rate). That baseline -- not "tokens saved" -- is the
354
+ correct comparison, because caching never reduces how many
355
+ tokens the model processes; see the module docstring and
356
+ percent_of_input_from_cache above.
357
+
358
+ This can come back NEGATIVE: a cache-WRITE call (the first call
359
+ with a new stable prefix) costs MORE than not caching at all,
360
+ since Anthropic charges a premium (1.25x for a 5-minute TTL,
361
+ 2x for 1-hour) to populate the cache. The saving only shows up
362
+ on a later cache-READ call against that same prefix. Report
363
+ both calls, not just the read, if you want an honest before/
364
+ after picture -- see examples/cache_savings_demo_anthropic.py (or the OpenAI/Gemini/generic siblings).
365
+
366
+ Uses tonst's own CACHE_READ/WRITE multiplier constants above
367
+ (verified against Anthropic's published pricing docs, not this
368
+ library's own guess), applied to the REAL token counts parsed
369
+ from an actual API response -- not an estimate.
370
+ """
371
+ read_mult = _cache_read_multiplier(model)
372
+ write_mult = CACHE_WRITE_MULTIPLIER_1H if cache_ttl == "1h" else CACHE_WRITE_MULTIPLIER_5M
373
+
374
+ actual_cost = (
375
+ self.input_tokens * 1.0
376
+ + self.cache_creation_input_tokens * write_mult
377
+ + self.cache_read_input_tokens * read_mult
378
+ )
379
+ total_tokens = (
380
+ self.input_tokens
381
+ + self.cache_creation_input_tokens
382
+ + self.cache_read_input_tokens
383
+ )
384
+ no_cache_cost = total_tokens * 1.0 # same token count, all at the base rate
385
+
386
+ if no_cache_cost == 0:
387
+ return 0.0
388
+ return round(100 * (no_cache_cost - actual_cost) / no_cache_cost, 1)
389
+
390
+
391
+ def parse_anthropic_usage(response_json: dict) -> CacheUsageReport:
392
+ """
393
+ Extracts cache-relevant token counts from a real Anthropic Messages
394
+ API response. Pass the full decoded JSON body of the response (the
395
+ function reads its "usage" key) -- this is the only way to actually
396
+ confirm caching worked, as opposed to just building a
397
+ correctly-shaped request and assuming it did.
398
+ """
399
+ usage = response_json.get("usage", {})
400
+ return CacheUsageReport(
401
+ input_tokens=usage.get("input_tokens", 0),
402
+ output_tokens=usage.get("output_tokens", 0),
403
+ cache_creation_input_tokens=usage.get("cache_creation_input_tokens", 0),
404
+ cache_read_input_tokens=usage.get("cache_read_input_tokens", 0),
405
+ )