tonst 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tonst/__init__.py +81 -0
- tonst/__main__.py +43 -0
- tonst/adapters.py +133 -0
- tonst/cache_structuring.py +405 -0
- tonst/client.py +1089 -0
- tonst/compactor.py +1074 -0
- tonst/gliner_redact.py +322 -0
- tonst/local_model.py +177 -0
- tonst/names.py +253 -0
- tonst/ollama_util.py +57 -0
- tonst/placeholders.py +254 -0
- tonst/providers/__init__.py +37 -0
- tonst/providers/gemini.py +335 -0
- tonst/providers/generic.py +225 -0
- tonst/providers/openai.py +219 -0
- tonst/providers/presets.py +91 -0
- tonst/rag.py +174 -0
- tonst/redact.py +423 -0
- tonst/redact_llm.py +401 -0
- tonst/relevance.py +134 -0
- tonst/savings_log.py +388 -0
- tonst/summarizers.py +221 -0
- tonst/token_count.py +158 -0
- tonst/tool_optimizer.py +451 -0
- tonst/trim.py +75 -0
- tonst-0.2.0.dist-info/METADATA +640 -0
- tonst-0.2.0.dist-info/RECORD +31 -0
- tonst-0.2.0.dist-info/WHEEL +5 -0
- tonst-0.2.0.dist-info/entry_points.txt +2 -0
- tonst-0.2.0.dist-info/licenses/LICENSE +21 -0
- tonst-0.2.0.dist-info/top_level.txt +1 -0
tonst/__init__.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
from .client import TonstClient, OptimizationReport, StructuredRedactionResult
|
|
2
|
+
from .cache_structuring import (
|
|
3
|
+
PromptParts,
|
|
4
|
+
structure_for_caching,
|
|
5
|
+
build_anthropic_cache_request,
|
|
6
|
+
check_cache_eligibility,
|
|
7
|
+
CacheEligibility,
|
|
8
|
+
parse_anthropic_usage,
|
|
9
|
+
CacheUsageReport,
|
|
10
|
+
)
|
|
11
|
+
from .compactor import (
|
|
12
|
+
HistoryCompactor,
|
|
13
|
+
compact_history,
|
|
14
|
+
CompactionResult,
|
|
15
|
+
compact_history_rolling,
|
|
16
|
+
RollingSummary,
|
|
17
|
+
RollingCompactionResult,
|
|
18
|
+
FoldJob,
|
|
19
|
+
run_fold_job,
|
|
20
|
+
estimate_fold_payback,
|
|
21
|
+
)
|
|
22
|
+
from .trim import flatten_messages
|
|
23
|
+
from .tool_optimizer import (
|
|
24
|
+
select_tools,
|
|
25
|
+
ToolSelection,
|
|
26
|
+
ToolSession,
|
|
27
|
+
build_anthropic_deferred_tools,
|
|
28
|
+
estimate_tool_tokens,
|
|
29
|
+
DEFERRED_TOOLS_SYSTEM_HINT,
|
|
30
|
+
)
|
|
31
|
+
from .rag import optimize_chunks, ChunkSelection
|
|
32
|
+
from .token_count import AnthropicTokenCounter, GeminiTokenCounter
|
|
33
|
+
from .summarizers import AnthropicSummarizer, GeminiSummarizer
|
|
34
|
+
from .savings_log import SavingsLog, summarize as summarize_savings, SavingsSummary
|
|
35
|
+
from .placeholders import PlaceholderFactory
|
|
36
|
+
from .redact import StreamRestorer, restore_placeholders
|
|
37
|
+
from . import providers
|
|
38
|
+
from . import adapters
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"PlaceholderFactory",
|
|
42
|
+
"restore_placeholders",
|
|
43
|
+
"StreamRestorer",
|
|
44
|
+
"TonstClient",
|
|
45
|
+
"OptimizationReport",
|
|
46
|
+
"StructuredRedactionResult",
|
|
47
|
+
"PromptParts",
|
|
48
|
+
"structure_for_caching",
|
|
49
|
+
"build_anthropic_cache_request",
|
|
50
|
+
"check_cache_eligibility",
|
|
51
|
+
"CacheEligibility",
|
|
52
|
+
"parse_anthropic_usage",
|
|
53
|
+
"CacheUsageReport",
|
|
54
|
+
"HistoryCompactor",
|
|
55
|
+
"compact_history",
|
|
56
|
+
"CompactionResult",
|
|
57
|
+
"compact_history_rolling",
|
|
58
|
+
"RollingSummary",
|
|
59
|
+
"RollingCompactionResult",
|
|
60
|
+
"FoldJob",
|
|
61
|
+
"run_fold_job",
|
|
62
|
+
"estimate_fold_payback",
|
|
63
|
+
"flatten_messages",
|
|
64
|
+
"select_tools",
|
|
65
|
+
"ToolSelection",
|
|
66
|
+
"ToolSession",
|
|
67
|
+
"build_anthropic_deferred_tools",
|
|
68
|
+
"estimate_tool_tokens",
|
|
69
|
+
"DEFERRED_TOOLS_SYSTEM_HINT",
|
|
70
|
+
"optimize_chunks",
|
|
71
|
+
"ChunkSelection",
|
|
72
|
+
"AnthropicTokenCounter",
|
|
73
|
+
"GeminiTokenCounter",
|
|
74
|
+
"AnthropicSummarizer",
|
|
75
|
+
"GeminiSummarizer",
|
|
76
|
+
"SavingsLog",
|
|
77
|
+
"summarize_savings",
|
|
78
|
+
"SavingsSummary",
|
|
79
|
+
"providers",
|
|
80
|
+
"adapters",
|
|
81
|
+
]
|
tonst/__main__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line entry point.
|
|
3
|
+
|
|
4
|
+
tonst stats [--log PATH] [--app NAME] [--since YYYY-MM-DD] [--json]
|
|
5
|
+
|
|
6
|
+
Also runnable without installing the console script:
|
|
7
|
+
|
|
8
|
+
python -m tonst stats
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
from .savings_log import summarize, format_summary
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def main(argv=None) -> int:
|
|
20
|
+
parser = argparse.ArgumentParser(prog="tonst", description="tonst command-line tools")
|
|
21
|
+
sub = parser.add_subparsers(dest="command")
|
|
22
|
+
|
|
23
|
+
stats = sub.add_parser("stats", help="summarize the local savings log")
|
|
24
|
+
stats.add_argument("--log", default=None, help="log path (default: $TONST_SAVINGS_LOG or ~/.tonst/savings.jsonl)")
|
|
25
|
+
stats.add_argument("--app", default=None, help="only include entries for this app name")
|
|
26
|
+
stats.add_argument("--since", default=None, help="only include entries on/after this ISO date, e.g. 2026-09-01")
|
|
27
|
+
stats.add_argument("--json", action="store_true", help="print machine-readable JSON")
|
|
28
|
+
|
|
29
|
+
args = parser.parse_args(argv)
|
|
30
|
+
if args.command != "stats":
|
|
31
|
+
parser.print_help()
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
summary = summarize(args.log, app=args.app, since=args.since)
|
|
35
|
+
if args.json:
|
|
36
|
+
print(json.dumps(summary.to_dict(), indent=2))
|
|
37
|
+
else:
|
|
38
|
+
print(format_summary(summary))
|
|
39
|
+
return 0
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
if __name__ == "__main__":
|
|
43
|
+
sys.exit(main())
|
tonst/adapters.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""
|
|
2
|
+
adapters.py
|
|
3
|
+
-----------
|
|
4
|
+
Small converters between tonst's plain message list and each provider's
|
|
5
|
+
request/response shapes, for TonstClient(messages_fn=...).
|
|
6
|
+
|
|
7
|
+
tonst hands messages_fn the final, redacted conversation as
|
|
8
|
+
[{"role": "system" | "user" | "assistant", "content": str}, ...].
|
|
9
|
+
These helpers turn that into what each API expects, and turn the API's
|
|
10
|
+
usage block back into the {"prompt_tokens", "cached_tokens"} dict tonst
|
|
11
|
+
reads (so the report and cache-aware compaction see real numbers).
|
|
12
|
+
|
|
13
|
+
import anthropic
|
|
14
|
+
from tonst import TonstClient
|
|
15
|
+
from tonst.adapters import to_anthropic, usage_from_anthropic
|
|
16
|
+
|
|
17
|
+
api = anthropic.Anthropic()
|
|
18
|
+
|
|
19
|
+
def call_claude(messages):
|
|
20
|
+
resp = api.messages.create(model="claude-sonnet-4-6", max_tokens=1024,
|
|
21
|
+
**to_anthropic(messages))
|
|
22
|
+
return resp.content[0].text, usage_from_anthropic(resp)
|
|
23
|
+
|
|
24
|
+
client = TonstClient(messages_fn=call_claude)
|
|
25
|
+
|
|
26
|
+
Every helper accepts either a plain dict (raw JSON) or an SDK response
|
|
27
|
+
object, and none of them imports a provider SDK.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
from typing import Optional
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _get(obj, *path, default=None):
|
|
35
|
+
"""obj.a.b or obj["a"]["b"], whichever exists."""
|
|
36
|
+
cur = obj
|
|
37
|
+
for key in path:
|
|
38
|
+
if cur is None:
|
|
39
|
+
return default
|
|
40
|
+
if isinstance(cur, dict):
|
|
41
|
+
cur = cur.get(key)
|
|
42
|
+
else:
|
|
43
|
+
cur = getattr(cur, key, None)
|
|
44
|
+
return default if cur is None else cur
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _merge_turns(messages: list, assistant_role: str) -> tuple:
|
|
48
|
+
"""(system text, [(role, [texts])]) with same-role neighbours merged and a user turn first."""
|
|
49
|
+
system = "\n\n".join(m["content"] for m in messages if m.get("role") == "system" and m.get("content"))
|
|
50
|
+
turns = []
|
|
51
|
+
for m in messages:
|
|
52
|
+
role = m.get("role")
|
|
53
|
+
if role == "system" or not m.get("content"):
|
|
54
|
+
continue
|
|
55
|
+
role = assistant_role if role == "assistant" else "user"
|
|
56
|
+
if turns and turns[-1][0] == role:
|
|
57
|
+
turns[-1][1].append(m["content"])
|
|
58
|
+
else:
|
|
59
|
+
turns.append((role, [m["content"]]))
|
|
60
|
+
if turns and turns[0][0] != "user":
|
|
61
|
+
turns.insert(0, ("user", ["(Earlier conversation omitted.)"]))
|
|
62
|
+
return system, turns
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def to_anthropic(messages: list, cache: bool = True) -> dict:
|
|
66
|
+
"""
|
|
67
|
+
Keyword arguments for Anthropic's Messages API: {"system", "messages"}.
|
|
68
|
+
Roles must alternate there, so same-role neighbours are merged (tonst's
|
|
69
|
+
summary message is a user turn and can sit next to one). cache=True puts
|
|
70
|
+
a cache_control breakpoint on the system prompt and a moving one on the
|
|
71
|
+
last block, which is how a multi-turn chat gets cached turn after turn.
|
|
72
|
+
Prompts under the model's minimum cacheable length are simply not cached.
|
|
73
|
+
"""
|
|
74
|
+
system, turns = _merge_turns(messages, "assistant")
|
|
75
|
+
out_msgs = [{"role": r, "content": [{"type": "text", "text": t} for t in texts]} for r, texts in turns]
|
|
76
|
+
if cache and out_msgs:
|
|
77
|
+
last = out_msgs[-1]["content"][-1]
|
|
78
|
+
out_msgs[-1]["content"][-1] = {**last, "cache_control": {"type": "ephemeral"}}
|
|
79
|
+
body = {"messages": out_msgs}
|
|
80
|
+
if system:
|
|
81
|
+
block = {"type": "text", "text": system}
|
|
82
|
+
if cache:
|
|
83
|
+
block["cache_control"] = {"type": "ephemeral"}
|
|
84
|
+
body["system"] = [block]
|
|
85
|
+
return body
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def to_openai(messages: list) -> list:
|
|
89
|
+
"""Chat Completions `messages`. OpenAI caches repeated prefixes automatically; no markers needed."""
|
|
90
|
+
return [{"role": m["role"], "content": m["content"]} for m in messages if m.get("content")]
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def to_gemini(messages: list) -> dict:
|
|
94
|
+
"""
|
|
95
|
+
generateContent REST body fields: {"systemInstruction", "contents"}
|
|
96
|
+
(role "model" for the assistant). Gemini 2.5+ caches repeated prefixes
|
|
97
|
+
implicitly; it's best-effort and needs a few thousand tokens.
|
|
98
|
+
"""
|
|
99
|
+
system, turns = _merge_turns(messages, "model")
|
|
100
|
+
body = {"contents": [{"role": r, "parts": [{"text": t} for t in texts]} for r, texts in turns]}
|
|
101
|
+
if system:
|
|
102
|
+
body["systemInstruction"] = {"parts": [{"text": system}]}
|
|
103
|
+
return body
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def usage_from_anthropic(resp) -> Optional[dict]:
|
|
107
|
+
u = _get(resp, "usage")
|
|
108
|
+
if u is None:
|
|
109
|
+
return None
|
|
110
|
+
read = int(_get(u, "cache_read_input_tokens", default=0))
|
|
111
|
+
total = int(_get(u, "input_tokens", default=0)) + int(_get(u, "cache_creation_input_tokens", default=0)) + read
|
|
112
|
+
return {"prompt_tokens": total, "cached_tokens": read}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def usage_from_openai(resp) -> Optional[dict]:
|
|
116
|
+
u = _get(resp, "usage")
|
|
117
|
+
if u is None:
|
|
118
|
+
return None
|
|
119
|
+
if _get(u, "prompt_tokens") is not None: # Chat Completions
|
|
120
|
+
return {"prompt_tokens": int(_get(u, "prompt_tokens")),
|
|
121
|
+
"cached_tokens": int(_get(u, "prompt_tokens_details", "cached_tokens", default=0))}
|
|
122
|
+
return {"prompt_tokens": int(_get(u, "input_tokens", default=0)), # Responses API
|
|
123
|
+
"cached_tokens": int(_get(u, "input_tokens_details", "cached_tokens", default=0))}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def usage_from_gemini(resp) -> Optional[dict]:
|
|
127
|
+
u = _get(resp, "usageMetadata") or _get(resp, "usage_metadata")
|
|
128
|
+
if u is None:
|
|
129
|
+
return None
|
|
130
|
+
prompt = _get(u, "promptTokenCount") if _get(u, "promptTokenCount") is not None else _get(u, "prompt_token_count", default=0)
|
|
131
|
+
cached = _get(u, "cachedContentTokenCount") if _get(u, "cachedContentTokenCount") is not None \
|
|
132
|
+
else _get(u, "cached_content_token_count", default=0)
|
|
133
|
+
return {"prompt_tokens": int(prompt), "cached_tokens": int(cached or 0)}
|
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
"""
|
|
2
|
+
cache_structuring.py
|
|
3
|
+
---------------------
|
|
4
|
+
Structures a request so that provider-side prompt caching actually pays
|
|
5
|
+
off. Important to be precise about what this is NOT: prompt caching does
|
|
6
|
+
not skip the LLM call. Anthropic, OpenAI, and Gemini all cache the
|
|
7
|
+
*processed representation* of repeated prefix tokens so the model
|
|
8
|
+
doesn't have to reprocess them, but the model still runs and generates a
|
|
9
|
+
fresh response on every call -- caching only discounts (and speeds up)
|
|
10
|
+
the repeated portion of the input.
|
|
11
|
+
|
|
12
|
+
It only pays off if the request is shaped correctly:
|
|
13
|
+
- Stable, reused content (system instructions, a reference document,
|
|
14
|
+
tool descriptions written out as text) has to come FIRST.
|
|
15
|
+
- That stable content has to be byte-for-byte IDENTICAL across calls --
|
|
16
|
+
reordering it, changing its whitespace, or interleaving the variable
|
|
17
|
+
question into it all silently break the match. There's no error when
|
|
18
|
+
this happens; you just quietly stop getting the discount.
|
|
19
|
+
- The variable part (the actual question, the newest turn) goes LAST.
|
|
20
|
+
- The stable prefix also has to clear a PER-MODEL MINIMUM LENGTH --
|
|
21
|
+
see CACHE_MINIMUM_TOKENS below. Below it, Anthropic silently skips
|
|
22
|
+
caching entirely: no error, cache_creation_input_tokens and
|
|
23
|
+
cache_read_input_tokens both stay 0. check_cache_eligibility() and
|
|
24
|
+
the warning in build_anthropic_cache_request() exist specifically so
|
|
25
|
+
this doesn't fail silently on you too.
|
|
26
|
+
|
|
27
|
+
Output shapes provided:
|
|
28
|
+
- structure_for_caching() returns a flat, correctly-ordered string.
|
|
29
|
+
This is enough for providers with automatic prefix caching (no
|
|
30
|
+
explicit marker required) and works with TonstClient.query(str).
|
|
31
|
+
- build_anthropic_cache_request() returns the actual Anthropic
|
|
32
|
+
Messages API JSON body with an explicit `cache_control` breakpoint,
|
|
33
|
+
which Anthropic requires -- correct ordering alone is not sufficient
|
|
34
|
+
on that provider. Verified against Anthropic's prompt-caching docs
|
|
35
|
+
(platform.claude.com/docs/en/build-with-claude/prompt-caching,
|
|
36
|
+
checked 2026-09-08): cache_control goes on the LAST block whose
|
|
37
|
+
prefix should be cached, as {"type": "ephemeral", "ttl": "5m"|"1h"}.
|
|
38
|
+
- parse_anthropic_usage() reads the `usage` field of a REAL API
|
|
39
|
+
response and reports whether a cache hit actually happened --
|
|
40
|
+
the only ground truth for whether any of this worked.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
import warnings
|
|
45
|
+
from dataclasses import dataclass, field
|
|
46
|
+
from typing import Optional
|
|
47
|
+
|
|
48
|
+
VALID_TTLS = ("5m", "1h")
|
|
49
|
+
|
|
50
|
+
# Anthropic's minimum cacheable-prefix length per model, in tokens, as of
|
|
51
|
+
# the docs checked 2026-09-08. A stable prefix shorter than this is
|
|
52
|
+
# processed normally with NO ERROR and NO CACHING -- the response's
|
|
53
|
+
# usage.cache_creation_input_tokens / usage.cache_read_input_tokens both
|
|
54
|
+
# come back 0. This table exists so tonst can warn about that ahead of
|
|
55
|
+
# time instead of a developer discovering it by silently getting no
|
|
56
|
+
# savings. Keyed on the model string as passed to the API; unknown/future
|
|
57
|
+
# models fall back to DEFAULT_CACHE_MINIMUM_TOKENS (the most common
|
|
58
|
+
# figure across the current lineup).
|
|
59
|
+
CACHE_MINIMUM_TOKENS: dict[str, int] = {
|
|
60
|
+
"claude-haiku-4-5": 4096,
|
|
61
|
+
"claude-haiku-3-5": 2048,
|
|
62
|
+
"claude-opus-4-6": 4096,
|
|
63
|
+
"claude-opus-4-5": 4096,
|
|
64
|
+
"claude-opus-4-7": 2048,
|
|
65
|
+
"claude-opus-4-8": 1024,
|
|
66
|
+
"claude-opus-4-1": 1024,
|
|
67
|
+
"claude-opus-4": 1024,
|
|
68
|
+
"claude-sonnet-5": 1024,
|
|
69
|
+
"claude-sonnet-4-6": 1024,
|
|
70
|
+
"claude-sonnet-4-5": 1024,
|
|
71
|
+
"claude-sonnet-4": 1024,
|
|
72
|
+
"claude-opus-5": 512,
|
|
73
|
+
"claude-fable-5": 512,
|
|
74
|
+
"claude-fable-5-1": 512,
|
|
75
|
+
"claude-mythos-5": 512,
|
|
76
|
+
"claude-mythos-5-1": 512,
|
|
77
|
+
# "Claude Mythos Preview" is documented at 2,048 tokens, but its
|
|
78
|
+
# exact API model-id string wasn't confirmed at the time this table
|
|
79
|
+
# was written -- add it here once known, rather than guess a key
|
|
80
|
+
# that would silently never match.
|
|
81
|
+
}
|
|
82
|
+
DEFAULT_CACHE_MINIMUM_TOKENS = 1024
|
|
83
|
+
|
|
84
|
+
# Anthropic's cache pricing multipliers, relative to that model's own
|
|
85
|
+
# base (uncached) input-token price of 1.0x. Verified against
|
|
86
|
+
# platform.claude.com/docs/en/build-with-claude/prompt-caching, checked
|
|
87
|
+
# 2026-09-09. A cache WRITE costs MORE than a fresh token -- you're
|
|
88
|
+
# paying a premium to populate the cache -- and only a cache READ is
|
|
89
|
+
# discounted. This is what makes "percent of input from cache" (see
|
|
90
|
+
# CacheUsageReport.percent_of_input_from_cache) different from an actual
|
|
91
|
+
# cost saving: caching never changes how many tokens get processed, so
|
|
92
|
+
# a real savings estimate has to weight each token type by its actual
|
|
93
|
+
# price, not just count what fraction came from cache.
|
|
94
|
+
CACHE_READ_MULTIPLIER_DEFAULT = 0.1 # 10% of base input price
|
|
95
|
+
CACHE_READ_MULTIPLIER_LOW_COST = 0.025 # Fable/Mythos family: 2.5% of base price
|
|
96
|
+
CACHE_WRITE_MULTIPLIER_5M = 1.25 # 125% of base input price
|
|
97
|
+
CACHE_WRITE_MULTIPLIER_1H = 2.0 # 200% of base input price
|
|
98
|
+
|
|
99
|
+
# Models with the lower 2.5% cache-read rate instead of the standard 10%.
|
|
100
|
+
_LOW_COST_CACHE_READ_MODELS = {
|
|
101
|
+
"claude-fable-5",
|
|
102
|
+
"claude-fable-5-1",
|
|
103
|
+
"claude-mythos-5",
|
|
104
|
+
"claude-mythos-5-1",
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _cache_read_multiplier(model: str) -> float:
|
|
109
|
+
return (
|
|
110
|
+
CACHE_READ_MULTIPLIER_LOW_COST
|
|
111
|
+
if model in _LOW_COST_CACHE_READ_MODELS
|
|
112
|
+
else CACHE_READ_MULTIPLIER_DEFAULT
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class PromptParts:
|
|
118
|
+
"""
|
|
119
|
+
system: instructions that never change between calls in this
|
|
120
|
+
conversation/session (e.g. "You are a support assistant...").
|
|
121
|
+
stable_blocks: large reusable content that's IDENTICAL across many
|
|
122
|
+
calls -- reference documents, few-shot examples, tool
|
|
123
|
+
descriptions written out as text. List order is preserved; put
|
|
124
|
+
the content least likely to ever change first, since everything
|
|
125
|
+
up to and including the cache_control breakpoint must match
|
|
126
|
+
exactly for a hit.
|
|
127
|
+
variable: the part that's different on every call -- the user's
|
|
128
|
+
actual question, or the newest turn. Always placed last.
|
|
129
|
+
"""
|
|
130
|
+
system: Optional[str] = None
|
|
131
|
+
stable_blocks: list = field(default_factory=list)
|
|
132
|
+
variable: str = ""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def structure_for_caching(parts: PromptParts) -> str:
|
|
136
|
+
"""
|
|
137
|
+
Returns a single string with stable content first, variable last.
|
|
138
|
+
Byte-for-byte identical stable content across calls is what lets
|
|
139
|
+
providers with automatic prefix caching (no explicit marker needed)
|
|
140
|
+
actually get a cache hit -- reordering, whitespace changes, or
|
|
141
|
+
interleaving the question into the stable section all silently
|
|
142
|
+
defeat it. This is the ordering guarantee; it does not, by itself,
|
|
143
|
+
set an explicit cache breakpoint (see build_anthropic_cache_request
|
|
144
|
+
for that).
|
|
145
|
+
"""
|
|
146
|
+
sections = []
|
|
147
|
+
if parts.system:
|
|
148
|
+
sections.append(parts.system.strip())
|
|
149
|
+
sections.extend(b.strip() for b in parts.stable_blocks if b and b.strip())
|
|
150
|
+
if parts.variable:
|
|
151
|
+
sections.append(parts.variable.strip())
|
|
152
|
+
return "\n\n".join(sections)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@dataclass
|
|
156
|
+
class CacheEligibility:
|
|
157
|
+
"""
|
|
158
|
+
Heuristic estimate of whether `parts`'s stable content is long enough
|
|
159
|
+
for `model` to actually cache it. This uses tonst's chars/4 token
|
|
160
|
+
estimate (see trim.py), NOT the provider's real tokenizer -- treat it
|
|
161
|
+
as a smoke test that catches the obvious "way too short" case, not
|
|
162
|
+
an exact prediction. The only real ground truth is the `usage` field
|
|
163
|
+
of an actual API response (see parse_anthropic_usage()).
|
|
164
|
+
"""
|
|
165
|
+
eligible: bool
|
|
166
|
+
stable_tokens_estimate: int
|
|
167
|
+
minimum_required: int
|
|
168
|
+
model: str
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def message(self) -> str:
|
|
172
|
+
if self.eligible:
|
|
173
|
+
return (
|
|
174
|
+
f"Stable content is ~{self.stable_tokens_estimate} tokens "
|
|
175
|
+
f"(est.), above {self.model}'s {self.minimum_required}-token "
|
|
176
|
+
"minimum -- likely eligible for caching."
|
|
177
|
+
)
|
|
178
|
+
return (
|
|
179
|
+
f"Stable content is only ~{self.stable_tokens_estimate} tokens "
|
|
180
|
+
f"(est.), below {self.model}'s {self.minimum_required}-token "
|
|
181
|
+
"minimum. This provider will NOT cache it: most providers raise "
|
|
182
|
+
"no error here, they simply process the request without "
|
|
183
|
+
"caching. Add more reusable content to stable_blocks/system, "
|
|
184
|
+
"or don't bother shaping this request for caching at all."
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def check_cache_eligibility(
|
|
189
|
+
parts: PromptParts, model: str, token_estimator=None, tools: Optional[list] = None
|
|
190
|
+
) -> CacheEligibility:
|
|
191
|
+
"""
|
|
192
|
+
Estimates whether the stable/cacheable portion of `parts` meets
|
|
193
|
+
Anthropic's minimum prefix length for `model`. Meant to catch an
|
|
194
|
+
obviously-too-short system prompt or reference snippet before
|
|
195
|
+
spending an API call that silently won't cache -- not a substitute
|
|
196
|
+
for checking real usage.cache_creation_input_tokens /
|
|
197
|
+
usage.cache_read_input_tokens from an actual response.
|
|
198
|
+
|
|
199
|
+
`tools`, if given, counts toward the cached prefix too (tool
|
|
200
|
+
definitions come first in Anthropic's prefix order) -- except tools
|
|
201
|
+
marked defer_loading, which the API keeps out of the prefix.
|
|
202
|
+
"""
|
|
203
|
+
if token_estimator is None:
|
|
204
|
+
from .trim import estimate_tokens as token_estimator
|
|
205
|
+
|
|
206
|
+
stable_text = "\n\n".join([parts.system or ""] + list(parts.stable_blocks))
|
|
207
|
+
stable_tokens = token_estimator(stable_text) if stable_text.strip() else 0
|
|
208
|
+
loaded = [t for t in (tools or []) if not t.get("defer_loading")]
|
|
209
|
+
if loaded:
|
|
210
|
+
import json
|
|
211
|
+
stable_tokens += token_estimator(json.dumps(loaded, separators=(",", ":"), sort_keys=True))
|
|
212
|
+
|
|
213
|
+
minimum = CACHE_MINIMUM_TOKENS.get(model, DEFAULT_CACHE_MINIMUM_TOKENS)
|
|
214
|
+
return CacheEligibility(
|
|
215
|
+
eligible=stable_tokens >= minimum,
|
|
216
|
+
stable_tokens_estimate=stable_tokens,
|
|
217
|
+
minimum_required=minimum,
|
|
218
|
+
model=model,
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def build_anthropic_cache_request(
|
|
223
|
+
parts: PromptParts,
|
|
224
|
+
model: str,
|
|
225
|
+
max_tokens: int = 1000,
|
|
226
|
+
cache_ttl: str = "5m",
|
|
227
|
+
warn_if_ineligible: bool = True,
|
|
228
|
+
tools: Optional[list] = None,
|
|
229
|
+
) -> dict:
|
|
230
|
+
"""
|
|
231
|
+
Builds the JSON body for a direct call to
|
|
232
|
+
https://api.anthropic.com/v1/messages with an explicit cache
|
|
233
|
+
breakpoint placed after the stable content. Anthropic requires this
|
|
234
|
+
`cache_control` marker on the last stable block -- ordering alone
|
|
235
|
+
(unlike OpenAI/Gemini's automatic prefix caching) does not trigger a
|
|
236
|
+
cache hit on this provider.
|
|
237
|
+
|
|
238
|
+
If `warn_if_ineligible` is True (default), this checks stable-content
|
|
239
|
+
length against CACHE_MINIMUM_TOKENS and raises a UserWarning if it's
|
|
240
|
+
likely too short to actually be cached. The request is still built
|
|
241
|
+
and returned either way -- Anthropic itself doesn't error on a
|
|
242
|
+
too-short cache_control block, it just silently skips caching, and
|
|
243
|
+
tonst matches that (fail-soft), just with a visible warning instead
|
|
244
|
+
of silence.
|
|
245
|
+
|
|
246
|
+
If `parts.system` is set, it's cached too (as its own block with its
|
|
247
|
+
own breakpoint) since a large, reused system prompt is one of the
|
|
248
|
+
most common things worth caching.
|
|
249
|
+
|
|
250
|
+
If there are no stable_blocks, this falls back to a plain string
|
|
251
|
+
message body -- there's nothing to cache, so the block-array
|
|
252
|
+
overhead (and the cost of writing a cache entry with no reuse ahead
|
|
253
|
+
of it) isn't worth it.
|
|
254
|
+
|
|
255
|
+
`tools` (optional) is passed through as the request's `tools` array,
|
|
256
|
+
unmodified -- e.g. the output of tool_optimizer.select_tools() or
|
|
257
|
+
build_anthropic_deferred_tools(). Tools come FIRST in Anthropic's
|
|
258
|
+
cache prefix, so the system breakpoint already covers them; when
|
|
259
|
+
there's no system prompt, a breakpoint is put on the last
|
|
260
|
+
non-deferred tool instead so the tool definitions still get cached.
|
|
261
|
+
"""
|
|
262
|
+
if cache_ttl not in VALID_TTLS:
|
|
263
|
+
raise ValueError(f"cache_ttl must be one of {VALID_TTLS}, got {cache_ttl!r}")
|
|
264
|
+
|
|
265
|
+
if warn_if_ineligible:
|
|
266
|
+
eligibility = check_cache_eligibility(parts, model, tools=tools)
|
|
267
|
+
if not eligibility.eligible:
|
|
268
|
+
warnings.warn(eligibility.message, UserWarning, stacklevel=2)
|
|
269
|
+
|
|
270
|
+
body: dict = {"model": model, "max_tokens": max_tokens}
|
|
271
|
+
|
|
272
|
+
if tools:
|
|
273
|
+
for t in tools:
|
|
274
|
+
if t.get("defer_loading") and "cache_control" in t:
|
|
275
|
+
raise ValueError(
|
|
276
|
+
f"tool {t.get('name')!r} has both defer_loading and cache_control; "
|
|
277
|
+
"Anthropic rejects that with a 400"
|
|
278
|
+
)
|
|
279
|
+
body["tools"] = [dict(t) for t in tools]
|
|
280
|
+
if not parts.system:
|
|
281
|
+
loaded_idx = [i for i, t in enumerate(body["tools"]) if not t.get("defer_loading")]
|
|
282
|
+
if loaded_idx and not any("cache_control" in t for t in body["tools"]):
|
|
283
|
+
body["tools"][loaded_idx[-1]]["cache_control"] = {"type": "ephemeral", "ttl": cache_ttl}
|
|
284
|
+
|
|
285
|
+
if parts.system:
|
|
286
|
+
body["system"] = [
|
|
287
|
+
{
|
|
288
|
+
"type": "text",
|
|
289
|
+
"text": parts.system.strip(),
|
|
290
|
+
"cache_control": {"type": "ephemeral", "ttl": cache_ttl},
|
|
291
|
+
}
|
|
292
|
+
]
|
|
293
|
+
|
|
294
|
+
stable = [b.strip() for b in parts.stable_blocks if b and b.strip()]
|
|
295
|
+
variable_text = (parts.variable or "").strip()
|
|
296
|
+
|
|
297
|
+
if stable:
|
|
298
|
+
content_blocks = [{"type": "text", "text": b} for b in stable]
|
|
299
|
+
# The breakpoint goes on the LAST stable block: everything up to
|
|
300
|
+
# and including it becomes the cached prefix, so only one marker
|
|
301
|
+
# is needed even if there are many stable blocks.
|
|
302
|
+
content_blocks[-1]["cache_control"] = {"type": "ephemeral", "ttl": cache_ttl}
|
|
303
|
+
if variable_text:
|
|
304
|
+
content_blocks.append({"type": "text", "text": variable_text})
|
|
305
|
+
body["messages"] = [{"role": "user", "content": content_blocks}]
|
|
306
|
+
else:
|
|
307
|
+
# Nothing stable to cache -- a plain string is simpler and avoids
|
|
308
|
+
# block-array overhead for a one-off call.
|
|
309
|
+
body["messages"] = [{"role": "user", "content": variable_text}]
|
|
310
|
+
|
|
311
|
+
return body
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
@dataclass
|
|
315
|
+
class CacheUsageReport:
|
|
316
|
+
"""
|
|
317
|
+
Parsed from a REAL Anthropic API response's `usage` field -- the
|
|
318
|
+
only ground truth for whether caching actually happened on a given
|
|
319
|
+
call. cache_creation_input_tokens > 0 means this call WROTE a new
|
|
320
|
+
cache entry (expected on the first call with a given stable prefix).
|
|
321
|
+
cache_read_input_tokens > 0 means this call GOT the discount by
|
|
322
|
+
reading a previously-written cache entry.
|
|
323
|
+
"""
|
|
324
|
+
input_tokens: int
|
|
325
|
+
output_tokens: int
|
|
326
|
+
cache_creation_input_tokens: int
|
|
327
|
+
cache_read_input_tokens: int
|
|
328
|
+
|
|
329
|
+
@property
|
|
330
|
+
def cache_hit(self) -> bool:
|
|
331
|
+
return self.cache_read_input_tokens > 0
|
|
332
|
+
|
|
333
|
+
@property
|
|
334
|
+
def cache_write(self) -> bool:
|
|
335
|
+
return self.cache_creation_input_tokens > 0
|
|
336
|
+
|
|
337
|
+
@property
|
|
338
|
+
def percent_of_input_from_cache(self) -> float:
|
|
339
|
+
total_input = (
|
|
340
|
+
self.input_tokens
|
|
341
|
+
+ self.cache_creation_input_tokens
|
|
342
|
+
+ self.cache_read_input_tokens
|
|
343
|
+
)
|
|
344
|
+
if total_input == 0:
|
|
345
|
+
return 0.0
|
|
346
|
+
return round(100 * self.cache_read_input_tokens / total_input, 1)
|
|
347
|
+
|
|
348
|
+
def estimated_cost_savings_percent(self, model: str, cache_ttl: str = "5m") -> float:
|
|
349
|
+
"""
|
|
350
|
+
Estimates the % COST difference caching made on this call,
|
|
351
|
+
against the baseline of the exact same token count being sent
|
|
352
|
+
with NO caching at all (everything billed at the standard 1x
|
|
353
|
+
input rate). That baseline -- not "tokens saved" -- is the
|
|
354
|
+
correct comparison, because caching never reduces how many
|
|
355
|
+
tokens the model processes; see the module docstring and
|
|
356
|
+
percent_of_input_from_cache above.
|
|
357
|
+
|
|
358
|
+
This can come back NEGATIVE: a cache-WRITE call (the first call
|
|
359
|
+
with a new stable prefix) costs MORE than not caching at all,
|
|
360
|
+
since Anthropic charges a premium (1.25x for a 5-minute TTL,
|
|
361
|
+
2x for 1-hour) to populate the cache. The saving only shows up
|
|
362
|
+
on a later cache-READ call against that same prefix. Report
|
|
363
|
+
both calls, not just the read, if you want an honest before/
|
|
364
|
+
after picture -- see examples/cache_savings_demo_anthropic.py (or the OpenAI/Gemini/generic siblings).
|
|
365
|
+
|
|
366
|
+
Uses tonst's own CACHE_READ/WRITE multiplier constants above
|
|
367
|
+
(verified against Anthropic's published pricing docs, not this
|
|
368
|
+
library's own guess), applied to the REAL token counts parsed
|
|
369
|
+
from an actual API response -- not an estimate.
|
|
370
|
+
"""
|
|
371
|
+
read_mult = _cache_read_multiplier(model)
|
|
372
|
+
write_mult = CACHE_WRITE_MULTIPLIER_1H if cache_ttl == "1h" else CACHE_WRITE_MULTIPLIER_5M
|
|
373
|
+
|
|
374
|
+
actual_cost = (
|
|
375
|
+
self.input_tokens * 1.0
|
|
376
|
+
+ self.cache_creation_input_tokens * write_mult
|
|
377
|
+
+ self.cache_read_input_tokens * read_mult
|
|
378
|
+
)
|
|
379
|
+
total_tokens = (
|
|
380
|
+
self.input_tokens
|
|
381
|
+
+ self.cache_creation_input_tokens
|
|
382
|
+
+ self.cache_read_input_tokens
|
|
383
|
+
)
|
|
384
|
+
no_cache_cost = total_tokens * 1.0 # same token count, all at the base rate
|
|
385
|
+
|
|
386
|
+
if no_cache_cost == 0:
|
|
387
|
+
return 0.0
|
|
388
|
+
return round(100 * (no_cache_cost - actual_cost) / no_cache_cost, 1)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def parse_anthropic_usage(response_json: dict) -> CacheUsageReport:
|
|
392
|
+
"""
|
|
393
|
+
Extracts cache-relevant token counts from a real Anthropic Messages
|
|
394
|
+
API response. Pass the full decoded JSON body of the response (the
|
|
395
|
+
function reads its "usage" key) -- this is the only way to actually
|
|
396
|
+
confirm caching worked, as opposed to just building a
|
|
397
|
+
correctly-shaped request and assuming it did.
|
|
398
|
+
"""
|
|
399
|
+
usage = response_json.get("usage", {})
|
|
400
|
+
return CacheUsageReport(
|
|
401
|
+
input_tokens=usage.get("input_tokens", 0),
|
|
402
|
+
output_tokens=usage.get("output_tokens", 0),
|
|
403
|
+
cache_creation_input_tokens=usage.get("cache_creation_input_tokens", 0),
|
|
404
|
+
cache_read_input_tokens=usage.get("cache_read_input_tokens", 0),
|
|
405
|
+
)
|