cachecanary 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cachecanary/__init__.py +3 -0
- cachecanary/cli.py +202 -0
- cachecanary/diff.py +91 -0
- cachecanary/github.py +50 -0
- cachecanary/lint.py +163 -0
- cachecanary/logs.py +129 -0
- cachecanary/models.py +77 -0
- cachecanary/probe.py +108 -0
- cachecanary/request.py +178 -0
- cachecanary/usage.py +96 -0
- cachecanary-0.1.0.dist-info/METADATA +153 -0
- cachecanary-0.1.0.dist-info/RECORD +17 -0
- cachecanary-0.1.0.dist-info/WHEEL +5 -0
- cachecanary-0.1.0.dist-info/entry_points.txt +2 -0
- cachecanary-0.1.0.dist-info/licenses/LICENSE +202 -0
- cachecanary-0.1.0.dist-info/licenses/NOTICE +4 -0
- cachecanary-0.1.0.dist-info/top_level.txt +1 -0
cachecanary/__init__.py
ADDED
cachecanary/cli.py
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""cachecanary command line.
|
|
2
|
+
|
|
3
|
+
cachecanary lint request.json [--model ID]
|
|
4
|
+
cachecanary diff previous.json next.json [--model ID]
|
|
5
|
+
cachecanary probe request.json --model ID [--region R] [--stream]
|
|
6
|
+
cachecanary logs file.json[.gz] ... [--by model|principal|model+principal] [--min-hit 0.5]
|
|
7
|
+
|
|
8
|
+
Exit codes: 0 ok, 1 caching problem found (use to gate CI), 2 could not run (bad input, AWS error).
|
|
9
|
+
Add --github (before the command) for GitHub Actions annotations and a job summary.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from dataclasses import asdict
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from cachecanary import __version__, diff, lint, logs, probe
|
|
19
|
+
from cachecanary import github as gh
|
|
20
|
+
from cachecanary.request import RequestError, normalize
|
|
21
|
+
|
|
22
|
+
EXIT_OK, EXIT_PROBLEM, EXIT_ERROR = 0, 1, 2
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class InputError(Exception):
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _load(path: str):
|
|
30
|
+
try:
|
|
31
|
+
return json.loads(Path(path).read_text(encoding="utf-8"))
|
|
32
|
+
except FileNotFoundError as exc:
|
|
33
|
+
raise InputError(f"file not found: {path}") from exc
|
|
34
|
+
except json.JSONDecodeError as exc:
|
|
35
|
+
raise InputError(f"{path} is not valid JSON: {exc}") from exc
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _emit(data, as_json: bool, lines: list[str]) -> None:
|
|
39
|
+
print(json.dumps(data, indent=2) if as_json else "\n".join(lines))
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def cmd_lint(args) -> int:
|
|
43
|
+
findings = lint.lint(normalize(_load(args.request), args.model))
|
|
44
|
+
lines = [f"[{f.severity}] {f.rule}: {f.message}" + (f" ({f.location})" if f.location else "") for f in findings]
|
|
45
|
+
_emit([asdict(f) for f in findings], args.json, lines or ["No caching problems found."])
|
|
46
|
+
if args.github:
|
|
47
|
+
for f in findings:
|
|
48
|
+
level = "error" if f.severity == "error" else "warning"
|
|
49
|
+
where = f" ({f.location})" if f.location else ""
|
|
50
|
+
print(gh.annotation(level, f"{f.message}{where}", file=args.request, title=f"CacheCanary: {f.rule}"))
|
|
51
|
+
rows = [[f.severity, f.rule, f.location or "", f.message] for f in findings]
|
|
52
|
+
gh.append_summary(f"### CacheCanary lint: `{args.request}`\n\n" + (
|
|
53
|
+
gh.table(["severity", "rule", "location", "message"], rows) if rows else "No caching problems found. ✅"))
|
|
54
|
+
return EXIT_PROBLEM if any(f.severity == "error" for f in findings) else EXIT_OK
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def cmd_diff(args) -> int:
|
|
58
|
+
reasons = diff.explain(normalize(_load(args.previous), args.model), normalize(_load(args.next), args.model))
|
|
59
|
+
lines = [f"{r.code}: {r.message}" + (f" ({r.location})" if r.location else "") for r in reasons]
|
|
60
|
+
_emit([asdict(r) for r in reasons], args.json, lines)
|
|
61
|
+
ok = bool(reasons) and reasons[0].code == "prefix-identical"
|
|
62
|
+
if args.github:
|
|
63
|
+
for r in reasons:
|
|
64
|
+
where = f" ({r.location})" if r.location else ""
|
|
65
|
+
print(gh.annotation("notice" if ok else "error", f"{r.message}{where}", file=args.next, title=f"CacheCanary: {r.code}"))
|
|
66
|
+
gh.append_summary(f"### CacheCanary diff: `{args.previous}` → `{args.next}`\n\n" +
|
|
67
|
+
gh.table(["reason", "location", "explanation"], [[r.code, r.location or "", r.message] for r in reasons]))
|
|
68
|
+
return EXIT_OK if ok else EXIT_PROBLEM
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _probe_explanation(result: probe.ProbeResult, model: str) -> list[str]:
|
|
72
|
+
first, second = result.first, result.second
|
|
73
|
+
if second is None:
|
|
74
|
+
return ["Could not read token usage from the response, so caching could not be verified."]
|
|
75
|
+
if result.passed:
|
|
76
|
+
return []
|
|
77
|
+
if first and first.cache_write == 0 and first.cache_read == 0:
|
|
78
|
+
return ["The first call wrote nothing to the cache: no checkpoint was applied (prefix below the "
|
|
79
|
+
"model minimum, no cache markers, or the library/model ID skipped caching)."]
|
|
80
|
+
if first and first.cache_write > 0:
|
|
81
|
+
hint = ["The first call wrote to the cache but the second did not read it."]
|
|
82
|
+
if model.split(".")[0] in ("us", "eu", "apac", "global", "jp", "au", "ca"):
|
|
83
|
+
hint.append("Cross-region profiles can route the two calls to different Regions; retry or test with "
|
|
84
|
+
"an in-Region model ID to rule this out.")
|
|
85
|
+
return hint
|
|
86
|
+
return []
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def cmd_probe(args) -> int:
|
|
90
|
+
result = probe.probe(_load(args.request), args.model, region=args.region, stream=args.stream)
|
|
91
|
+
data = {"passed": result.passed, "first": asdict(result.first) if result.first else None,
|
|
92
|
+
"second": asdict(result.second) if result.second else None}
|
|
93
|
+
status = "PASS: second call read from cache." if result.passed else "FAIL: second call did not read from cache."
|
|
94
|
+
lines = [status, f"first: {data['first']}", f"second: {data['second']}"]
|
|
95
|
+
lines += _probe_explanation(result, args.model)
|
|
96
|
+
if not result.passed:
|
|
97
|
+
lines.append("Run `cachecanary lint` on the same request to find the likely cause.")
|
|
98
|
+
_emit(data, args.json, lines)
|
|
99
|
+
if args.github:
|
|
100
|
+
if not result.passed:
|
|
101
|
+
print(gh.annotation("error", " ".join(lines[3:]) or status, file=args.request, title="CacheCanary: cache not read on 2nd call"))
|
|
102
|
+
rows = [[name, u["uncached_input"], u["cache_read"], u["cache_write"]] if u else [name, "?", "?", "?"]
|
|
103
|
+
for name, u in (("1st call", data["first"]), ("2nd call", data["second"]))]
|
|
104
|
+
gh.append_summary(f"### CacheCanary probe: `{args.request}` on `{args.model}` — {'PASS ✅' if result.passed else 'FAIL ❌'}\n\n"
|
|
105
|
+
+ gh.table(["call", "uncached", "cache read", "cache write"], rows))
|
|
106
|
+
return EXIT_OK if result.passed else EXIT_PROBLEM
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def cmd_logs(args) -> int:
|
|
110
|
+
stats = logs.ReadStats()
|
|
111
|
+
records = []
|
|
112
|
+
for p in args.files:
|
|
113
|
+
path = Path(p)
|
|
114
|
+
if not path.exists():
|
|
115
|
+
raise InputError(f"file not found: {p}")
|
|
116
|
+
try:
|
|
117
|
+
records.extend(logs.iter_records(path, stats))
|
|
118
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
119
|
+
raise InputError(f"could not read {p}: {exc}") from exc
|
|
120
|
+
groups = logs.aggregate(records, by=args.by)
|
|
121
|
+
data, lines, failing = {}, [], False
|
|
122
|
+
for key, g in sorted(groups.items(), key=lambda kv: -kv[1].usage.total_input):
|
|
123
|
+
ratio = g.usage.hit_ratio
|
|
124
|
+
data[key] = {"calls": g.calls, "unparsed": g.unparsed, "hit_ratio": round(ratio, 3), **asdict(g.usage)}
|
|
125
|
+
flag = ""
|
|
126
|
+
if args.min_hit is not None and g.calls - g.unparsed > 0 and ratio < args.min_hit:
|
|
127
|
+
flag, failing = " <-- below threshold", True
|
|
128
|
+
lines.append(f"{key}: hit {ratio:.0%} over {g.calls} calls "
|
|
129
|
+
f"(read {g.usage.cache_read}, write {g.usage.cache_write}, uncached {g.usage.uncached_input}, "
|
|
130
|
+
f"unparsed {g.unparsed}){flag}")
|
|
131
|
+
if stats.bad_lines:
|
|
132
|
+
lines.append(f"Skipped {stats.bad_lines} unreadable line(s).")
|
|
133
|
+
_emit(data, args.json, lines or ["No invocation log records found."])
|
|
134
|
+
if args.github:
|
|
135
|
+
rows = []
|
|
136
|
+
for key, d in data.items():
|
|
137
|
+
below = args.min_hit is not None and d["calls"] - d["unparsed"] > 0 and d["hit_ratio"] < args.min_hit
|
|
138
|
+
if below:
|
|
139
|
+
print(gh.annotation("error", f"{key}: cache hit rate {d['hit_ratio']:.0%} is below {args.min_hit:.0%}",
|
|
140
|
+
title="CacheCanary: low cache hit rate"))
|
|
141
|
+
rows.append([key, d["calls"], f"{d['hit_ratio']:.0%}", d["cache_read"], d["cache_write"], d["uncached_input"],
|
|
142
|
+
d["unparsed"], "❌" if below else ""])
|
|
143
|
+
gh.append_summary("### CacheCanary logs\n\n" + (gh.table(
|
|
144
|
+
["group", "calls", "hit rate", "read", "write", "uncached", "unparsed", "below threshold"], rows)
|
|
145
|
+
if rows else "No invocation log records found."))
|
|
146
|
+
return EXIT_PROBLEM if failing else EXIT_OK
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _unit(value: str) -> float:
|
|
150
|
+
f = float(value)
|
|
151
|
+
if not 0 <= f <= 1:
|
|
152
|
+
raise argparse.ArgumentTypeError("must be between 0 and 1")
|
|
153
|
+
return f
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
157
|
+
parser = argparse.ArgumentParser(prog="cachecanary", description="CacheCanary: catch silent prompt-cache breakage for Claude on Amazon Bedrock.")
|
|
158
|
+
parser.add_argument("--version", action="version", version=f"cachecanary {__version__}")
|
|
159
|
+
parser.add_argument("--json", action="store_true", help="machine-readable output")
|
|
160
|
+
parser.add_argument("--github", action="store_true",
|
|
161
|
+
help="also emit GitHub Actions annotations and a job summary (used by the GitHub Action)")
|
|
162
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
163
|
+
|
|
164
|
+
p = sub.add_parser("lint", help="static checks on one request")
|
|
165
|
+
p.add_argument("request")
|
|
166
|
+
p.add_argument("--model")
|
|
167
|
+
p.set_defaults(func=cmd_lint)
|
|
168
|
+
|
|
169
|
+
p = sub.add_parser("diff", help="explain why the next request missed the previous one's cache")
|
|
170
|
+
p.add_argument("previous")
|
|
171
|
+
p.add_argument("next")
|
|
172
|
+
p.add_argument("--model")
|
|
173
|
+
p.set_defaults(func=cmd_diff)
|
|
174
|
+
|
|
175
|
+
p = sub.add_parser("probe", help="send the request twice and check the second reads from cache")
|
|
176
|
+
p.add_argument("request")
|
|
177
|
+
p.add_argument("--model", required=True)
|
|
178
|
+
p.add_argument("--region")
|
|
179
|
+
p.add_argument("--stream", action="store_true", help="use ConverseStream / InvokeModelWithResponseStream")
|
|
180
|
+
p.set_defaults(func=cmd_probe)
|
|
181
|
+
|
|
182
|
+
p = sub.add_parser("logs", help="cache hit rates from Bedrock invocation logs")
|
|
183
|
+
p.add_argument("files", nargs="+")
|
|
184
|
+
p.add_argument("--by", choices=["model", "principal", "model+principal"], default="model")
|
|
185
|
+
p.add_argument("--min-hit", type=_unit, help="fail if any group's hit ratio is below this (0-1)")
|
|
186
|
+
p.set_defaults(func=cmd_logs)
|
|
187
|
+
return parser
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def main(argv: list[str] | None = None) -> int:
|
|
191
|
+
args = build_parser().parse_args(argv)
|
|
192
|
+
try:
|
|
193
|
+
return args.func(args)
|
|
194
|
+
except (InputError, RequestError, probe.ProbeError) as exc:
|
|
195
|
+
print(f"cachecanary: error: {exc}", file=sys.stderr)
|
|
196
|
+
if getattr(args, "github", False):
|
|
197
|
+
print(gh.annotation("error", str(exc), title="CacheCanary could not run"))
|
|
198
|
+
return EXIT_ERROR
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
if __name__ == "__main__":
|
|
202
|
+
sys.exit(main())
|
cachecanary/diff.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Explain why request B missed the cache that request A should have created.
|
|
2
|
+
|
|
3
|
+
This rebuilds, for Bedrock, what Anthropic's cache diagnostics reports on the Claude API:
|
|
4
|
+
compare consecutive requests and name the first thing that changed inside the cached prefix.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
from cachecanary.models import LOOKBACK_BLOCKS
|
|
10
|
+
from cachecanary.request import NormalizedRequest
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class MissReason:
|
|
15
|
+
code: str
|
|
16
|
+
message: str
|
|
17
|
+
location: str | None = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def explain(a: NormalizedRequest, b: NormalizedRequest) -> list[MissReason]:
|
|
21
|
+
reasons: list[MissReason] = []
|
|
22
|
+
|
|
23
|
+
if a.api != b.api:
|
|
24
|
+
reasons.append(MissReason(
|
|
25
|
+
"api-changed",
|
|
26
|
+
f"Requests switched API ({a.api} -> {b.api}). Converse and InvokeModel build different prompts, "
|
|
27
|
+
"so the cache entry cannot be reused (common after a gateway/library upgrade).",
|
|
28
|
+
))
|
|
29
|
+
return reasons
|
|
30
|
+
|
|
31
|
+
if (a.model_id or "") != (b.model_id or ""):
|
|
32
|
+
reasons.append(MissReason("model-changed", f"Model changed from {a.model_id} to {b.model_id}. Caches are per model."))
|
|
33
|
+
return reasons
|
|
34
|
+
|
|
35
|
+
a_cps = a.checkpoint_indexes
|
|
36
|
+
if not a_cps:
|
|
37
|
+
reasons.append(MissReason("no-checkpoint-in-previous", "The previous request had no checkpoint, so it wrote nothing to read back."))
|
|
38
|
+
return reasons
|
|
39
|
+
a_last = a_cps[-1]
|
|
40
|
+
|
|
41
|
+
if not b.checkpoint_indexes:
|
|
42
|
+
reasons.append(MissReason(
|
|
43
|
+
"no-checkpoint-in-next",
|
|
44
|
+
"The new request has no cache checkpoint, so it never asks to read the cache "
|
|
45
|
+
"(a library or code path stopped sending cache markers).",
|
|
46
|
+
))
|
|
47
|
+
return reasons
|
|
48
|
+
|
|
49
|
+
for i in range(min(a_last + 1, len(b.blocks))):
|
|
50
|
+
if a.blocks[i].content != b.blocks[i].content:
|
|
51
|
+
reasons.append(_classify_change(a, b, i))
|
|
52
|
+
return reasons
|
|
53
|
+
if len(b.blocks) <= a_last:
|
|
54
|
+
reasons.append(MissReason("prefix-truncated", "The new request is shorter than the previous cached prefix (history was trimmed or compacted)."))
|
|
55
|
+
return reasons
|
|
56
|
+
|
|
57
|
+
a_ttls = [a.blocks[i].ttl for i in a_cps]
|
|
58
|
+
b_ttls = [b.blocks[i].ttl for i in b.checkpoint_indexes]
|
|
59
|
+
if a_ttls and b_ttls and a_ttls[0] != b_ttls[0]:
|
|
60
|
+
reasons.append(MissReason("ttl-changed", f"Checkpoint TTL changed ({a_ttls[0]} -> {b_ttls[0]}); entries are keyed by TTL."))
|
|
61
|
+
|
|
62
|
+
b_cps = b.checkpoint_indexes
|
|
63
|
+
if b_cps and b_cps[-1] - a_last > LOOKBACK_BLOCKS and not any(a_last <= i < b_cps[-1] for i in b_cps[:-1]):
|
|
64
|
+
reasons.append(MissReason(
|
|
65
|
+
"lookback-exceeded",
|
|
66
|
+
f"{b_cps[-1] - a_last} blocks were added between the previous checkpoint and the new one. Bedrock only "
|
|
67
|
+
f"looks back ~{LOOKBACK_BLOCKS} blocks, so it cannot find the earlier cache entry. Add an intermediate "
|
|
68
|
+
"checkpoint (common after many parallel tool calls).",
|
|
69
|
+
b.blocks[b_cps[-1]].location,
|
|
70
|
+
))
|
|
71
|
+
|
|
72
|
+
if not reasons:
|
|
73
|
+
reasons.append(MissReason(
|
|
74
|
+
"prefix-identical",
|
|
75
|
+
"The cached prefix is identical. If B still missed, the entry likely expired (TTL elapsed between calls) "
|
|
76
|
+
"or cross-region inference routed to a Region without the entry.",
|
|
77
|
+
))
|
|
78
|
+
return reasons
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _classify_change(a: NormalizedRequest, b: NormalizedRequest, i: int) -> MissReason:
|
|
82
|
+
section = a.blocks[i].section
|
|
83
|
+
if section == "tools":
|
|
84
|
+
a_tools = sorted(x.content for x in a.blocks if x.section == "tools")
|
|
85
|
+
b_tools = sorted(x.content for x in b.blocks if x.section == "tools")
|
|
86
|
+
if a_tools == b_tools:
|
|
87
|
+
return MissReason("tools-reordered", "Same tools, different order. Sort tools deterministically.", b.blocks[i].location)
|
|
88
|
+
return MissReason("tools-changed", "A tool definition changed; tools come first, so the whole cache is invalidated.", b.blocks[i].location)
|
|
89
|
+
if section == "system":
|
|
90
|
+
return MissReason("system-changed", "The system prompt changed inside the cached prefix (look for dates, IDs or per-user text).", b.blocks[i].location)
|
|
91
|
+
return MissReason("history-changed", "Earlier conversation history was edited (not just appended).", b.blocks[i].location)
|
cachecanary/github.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""GitHub Actions output: inline annotations (workflow commands) and a job summary.
|
|
2
|
+
|
|
3
|
+
Annotations appear on the pull request / run page; the summary is Markdown appended to the
|
|
4
|
+
file named by $GITHUB_STEP_SUMMARY. Both are no-ops outside GitHub Actions apart from printing.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _escape_data(value: str) -> str:
|
|
11
|
+
# Per GitHub's workflow-command escaping rules.
|
|
12
|
+
return value.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _escape_property(value: str) -> str:
|
|
16
|
+
return _escape_data(value).replace(":", "%3A").replace(",", "%2C")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def annotation(level: str, message: str, file: str | None = None, title: str | None = None) -> str:
|
|
20
|
+
"""level: 'error' | 'warning' | 'notice'."""
|
|
21
|
+
if level not in ("error", "warning", "notice"):
|
|
22
|
+
raise ValueError(f"unknown annotation level: {level}")
|
|
23
|
+
props = []
|
|
24
|
+
if file:
|
|
25
|
+
props.append(f"file={_escape_property(file)}")
|
|
26
|
+
if title:
|
|
27
|
+
props.append(f"title={_escape_property(title)}")
|
|
28
|
+
head = f"::{level}" + (" " + ",".join(props) if props else "")
|
|
29
|
+
return f"{head}::{_escape_data(message)}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def append_summary(markdown: str) -> bool:
|
|
33
|
+
"""Append to the job summary. Returns False when not running in GitHub Actions."""
|
|
34
|
+
path = os.environ.get("GITHUB_STEP_SUMMARY")
|
|
35
|
+
if not path:
|
|
36
|
+
return False
|
|
37
|
+
with open(path, "a", encoding="utf-8") as fh:
|
|
38
|
+
fh.write(markdown.rstrip() + "\n\n")
|
|
39
|
+
return True
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def md_cell(value) -> str:
|
|
43
|
+
"""Make a value safe inside a Markdown table cell."""
|
|
44
|
+
return str(value).replace("|", "\\|").replace("\n", " ")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def table(headers: list[str], rows: list[list]) -> str:
|
|
48
|
+
out = ["| " + " | ".join(headers) + " |", "|" + "---|" * len(headers)]
|
|
49
|
+
out += ["| " + " | ".join(md_cell(c) for c in row) + " |" for row in rows]
|
|
50
|
+
return "\n".join(out)
|
cachecanary/lint.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Static checks on a single Bedrock request for patterns that silently defeat caching."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
from cachecanary.models import is_profile_arn, lookup
|
|
7
|
+
from cachecanary.request import NormalizedRequest, estimate_tokens
|
|
8
|
+
|
|
9
|
+
# Values that usually change per request. Matching them inside the cached prefix is
|
|
10
|
+
# the classic "cache never hits" bug (e.g. "Today's date: ..." in the system prompt).
|
|
11
|
+
DYNAMIC_PATTERNS = {
|
|
12
|
+
"ISO date": re.compile(r"\b20\d{2}-[01]\d-[0-3]\d\b"),
|
|
13
|
+
"clock time": re.compile(r"\b[0-2]?\d:[0-5]\d(:[0-5]\d)?\b"),
|
|
14
|
+
"UUID": re.compile(r"\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b", re.I),
|
|
15
|
+
"unix timestamp": re.compile(r"\b1[6-9]\d{8}(\d{3})?\b"),
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
# Conversation blocks after the last checkpoint before we suggest caching the history too.
|
|
19
|
+
MIN_UNCACHED_TAIL = 4
|
|
20
|
+
|
|
21
|
+
# Token estimates are approximate: live Bedrock runs (Oct 2026) counted ~11% MORE tokens than
|
|
22
|
+
# estimate_tokens() for English prompts. Only call a prefix "too short" when it is clearly below
|
|
23
|
+
# the minimum; inside the uncertainty band, warn and point to `probe` for a definitive answer.
|
|
24
|
+
CLEARLY_BELOW = 0.80
|
|
25
|
+
UNCERTAIN_UP_TO = 1.20
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Finding:
|
|
30
|
+
rule: str
|
|
31
|
+
severity: str # "error" = caching cannot work as written; "warn" = likely miss / money left on table
|
|
32
|
+
message: str
|
|
33
|
+
location: str | None = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def lint(req: NormalizedRequest) -> list[Finding]:
|
|
37
|
+
findings: list[Finding] = []
|
|
38
|
+
cps = req.checkpoint_indexes
|
|
39
|
+
key, limits = lookup(req.model_id)
|
|
40
|
+
|
|
41
|
+
if not req.model_id:
|
|
42
|
+
findings.append(Finding(
|
|
43
|
+
"no-model", "warn",
|
|
44
|
+
"No model ID given (use --model). Model-specific checks such as the minimum prefix size were skipped.",
|
|
45
|
+
))
|
|
46
|
+
elif is_profile_arn(req.model_id):
|
|
47
|
+
findings.append(Finding(
|
|
48
|
+
"profile-arn", "warn",
|
|
49
|
+
"Model is an application inference profile ARN. Many libraries match model names to "
|
|
50
|
+
"decide whether to send cache checkpoints and silently skip this case. Verify the "
|
|
51
|
+
"response shows cache reads/writes. Minimum-size checks were skipped (model unknown).",
|
|
52
|
+
))
|
|
53
|
+
elif key is None:
|
|
54
|
+
findings.append(Finding(
|
|
55
|
+
"unknown-model", "warn",
|
|
56
|
+
f"'{req.model_id}' is not in the known Claude caching table. New model IDs are a common "
|
|
57
|
+
"reason libraries stop adding checkpoints. Verify cache usage in responses.",
|
|
58
|
+
))
|
|
59
|
+
|
|
60
|
+
for loc in req.orphan_checkpoints:
|
|
61
|
+
findings.append(Finding(
|
|
62
|
+
"orphan-checkpoint", "warn", "Cache checkpoint with no content before it; it caches nothing.", loc,
|
|
63
|
+
))
|
|
64
|
+
|
|
65
|
+
if req.system_is_string and not any(req.blocks[i].section in ("tools", "system") for i in cps):
|
|
66
|
+
findings.append(Finding(
|
|
67
|
+
"string-system", "warn",
|
|
68
|
+
"'system' is a plain string, which cannot carry cache_control. Send it as a list of "
|
|
69
|
+
"text blocks and put cache_control on the last one to cache the system prompt.",
|
|
70
|
+
"system",
|
|
71
|
+
))
|
|
72
|
+
|
|
73
|
+
if not cps:
|
|
74
|
+
findings.append(Finding(
|
|
75
|
+
"no-checkpoint", "warn",
|
|
76
|
+
"No explicit cache checkpoint. Only best-effort implicit caching applies; add a "
|
|
77
|
+
"checkpoint after the static content (tools/system) for reliable hits.",
|
|
78
|
+
))
|
|
79
|
+
return findings
|
|
80
|
+
|
|
81
|
+
max_cp = limits.max_checkpoints if limits else 4
|
|
82
|
+
if req.checkpoint_markers > max_cp:
|
|
83
|
+
findings.append(Finding(
|
|
84
|
+
"too-many-checkpoints", "error",
|
|
85
|
+
f"{req.checkpoint_markers} cache markers; the maximum is {max_cp}. Bedrock rejects the request or ignores the extras.",
|
|
86
|
+
))
|
|
87
|
+
for loc in req.duplicate_checkpoints:
|
|
88
|
+
findings.append(Finding(
|
|
89
|
+
"duplicate-checkpoint", "warn", "Two cache markers in a row; the second adds nothing but counts toward the limit.", loc,
|
|
90
|
+
))
|
|
91
|
+
|
|
92
|
+
if limits:
|
|
93
|
+
cumulative = 0
|
|
94
|
+
clearly_short: list[tuple[int, int]] = []
|
|
95
|
+
borderline: list[tuple[int, int]] = []
|
|
96
|
+
cached_any = False
|
|
97
|
+
for i, block in enumerate(req.blocks[: cps[-1] + 1]):
|
|
98
|
+
cumulative += estimate_tokens(block.content)
|
|
99
|
+
if not block.checkpoint:
|
|
100
|
+
continue
|
|
101
|
+
if cumulative < limits.min_tokens * CLEARLY_BELOW:
|
|
102
|
+
clearly_short.append((i, cumulative))
|
|
103
|
+
elif cumulative < limits.min_tokens * UNCERTAIN_UP_TO:
|
|
104
|
+
borderline.append((i, cumulative))
|
|
105
|
+
cached_any = cached_any or cumulative >= limits.min_tokens
|
|
106
|
+
else:
|
|
107
|
+
cached_any = True
|
|
108
|
+
for i, prefix in clearly_short:
|
|
109
|
+
consequence = (
|
|
110
|
+
"The request succeeds but nothing is cached."
|
|
111
|
+
if not cached_any and not borderline
|
|
112
|
+
else "This checkpoint is ignored; later ones may still cache."
|
|
113
|
+
)
|
|
114
|
+
findings.append(Finding(
|
|
115
|
+
"prefix-too-short", "error" if not cached_any and not borderline else "warn",
|
|
116
|
+
f"Prefix up to this checkpoint is ~{prefix} tokens; {key} needs at least {limits.min_tokens}. {consequence}",
|
|
117
|
+
req.blocks[i].location,
|
|
118
|
+
))
|
|
119
|
+
for i, prefix in borderline:
|
|
120
|
+
findings.append(Finding(
|
|
121
|
+
"prefix-near-minimum", "warn",
|
|
122
|
+
f"Prefix up to this checkpoint is ~{prefix} tokens (estimate), close to {key}'s minimum of "
|
|
123
|
+
f"{limits.min_tokens}. Run `cachecanary probe` to confirm it is actually cached.",
|
|
124
|
+
req.blocks[i].location,
|
|
125
|
+
))
|
|
126
|
+
|
|
127
|
+
ttls = [req.blocks[i].ttl for i in cps]
|
|
128
|
+
unknown_ttls = sorted({t for t in ttls if t not in ("5m", "1h")})
|
|
129
|
+
if unknown_ttls:
|
|
130
|
+
findings.append(Finding(
|
|
131
|
+
"ttl-invalid", "error", f"Unsupported TTL value(s) {unknown_ttls}; only '5m' and '1h' exist.",
|
|
132
|
+
))
|
|
133
|
+
if limits and not limits.supports_1h and "1h" in ttls:
|
|
134
|
+
findings.append(Finding(
|
|
135
|
+
"ttl-unsupported", "error", f"{key} only supports the 5-minute TTL; 'ttl: 1h' can raise a ValidationException.",
|
|
136
|
+
))
|
|
137
|
+
if "5m" in ttls and "1h" in ttls and ttls.index("5m") < max(i for i, t in enumerate(ttls) if t == "1h"):
|
|
138
|
+
findings.append(Finding(
|
|
139
|
+
"ttl-order", "error", "A 1h checkpoint appears after a 5m checkpoint. Longer TTLs must come first.",
|
|
140
|
+
))
|
|
141
|
+
|
|
142
|
+
last = cps[-1]
|
|
143
|
+
for block in req.blocks[: last + 1]:
|
|
144
|
+
for label, pattern in DYNAMIC_PATTERNS.items():
|
|
145
|
+
if pattern.search(block.content):
|
|
146
|
+
findings.append(Finding(
|
|
147
|
+
"dynamic-in-prefix", "warn",
|
|
148
|
+
f"Found {label} text inside the cached prefix. If it changes per request, every call misses.",
|
|
149
|
+
block.location,
|
|
150
|
+
))
|
|
151
|
+
break
|
|
152
|
+
|
|
153
|
+
tail = len(req.blocks) - 1 - last
|
|
154
|
+
# Short tails are normal (one new question); only flag history that is re-billed every turn.
|
|
155
|
+
if tail >= MIN_UNCACHED_TAIL and req.blocks[last].section != "messages":
|
|
156
|
+
findings.append(Finding(
|
|
157
|
+
"no-conversation-checkpoint", "warn",
|
|
158
|
+
f"The last checkpoint is in {req.blocks[last].section}; {tail} conversation blocks after it are "
|
|
159
|
+
"re-billed every turn. Add a checkpoint near the end of the conversation history.",
|
|
160
|
+
req.blocks[last].location,
|
|
161
|
+
))
|
|
162
|
+
|
|
163
|
+
return findings
|
cachecanary/logs.py
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Aggregate cache hit rates from Bedrock model-invocation logs.
|
|
2
|
+
|
|
3
|
+
Invocation log records only carry inputTokenCount/outputTokenCount at the top level; the
|
|
4
|
+
cache counts live in output.outputBodyJson (the model response), which is inline only when
|
|
5
|
+
the body is <= 100 KB. Records whose body was offloaded to S3 are counted as 'unparsed'.
|
|
6
|
+
|
|
7
|
+
Accepted inputs (optionally .gz):
|
|
8
|
+
- S3 delivery: one JSON record per line, or a JSON array of records
|
|
9
|
+
- CloudWatch Logs export to S3: "<ISO timestamp> <json record>" per line
|
|
10
|
+
- `aws logs filter-log-events` output: {"events": [{"message": "<json record>"}, ...]}
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import gzip
|
|
14
|
+
import json
|
|
15
|
+
from collections import defaultdict
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from cachecanary.models import canonical_model_id
|
|
20
|
+
from cachecanary.usage import Usage, from_response
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Group:
|
|
25
|
+
calls: int = 0
|
|
26
|
+
unparsed: int = 0
|
|
27
|
+
usage: Usage = field(default_factory=Usage)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class ReadStats:
|
|
32
|
+
bad_lines: int = 0
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _iter_concatenated(text: str, stats: ReadStats | None):
|
|
36
|
+
"""Records glued together ('{...}{...}') or one per line; skips unreadable fragments."""
|
|
37
|
+
decoder = json.JSONDecoder()
|
|
38
|
+
i, n = 0, len(text)
|
|
39
|
+
while i < n:
|
|
40
|
+
while i < n and text[i].isspace():
|
|
41
|
+
i += 1
|
|
42
|
+
if i >= n:
|
|
43
|
+
return
|
|
44
|
+
if text[i] != "{":
|
|
45
|
+
# CloudWatch export: "<timestamp> {json}" -> skip to the next object start.
|
|
46
|
+
nxt = text.find("{", i)
|
|
47
|
+
if nxt == -1:
|
|
48
|
+
if stats:
|
|
49
|
+
stats.bad_lines += 1
|
|
50
|
+
return
|
|
51
|
+
i = nxt
|
|
52
|
+
try:
|
|
53
|
+
obj, end = decoder.raw_decode(text, i)
|
|
54
|
+
except json.JSONDecodeError:
|
|
55
|
+
if stats:
|
|
56
|
+
stats.bad_lines += 1
|
|
57
|
+
nl = text.find("\n", i)
|
|
58
|
+
if nl == -1:
|
|
59
|
+
return
|
|
60
|
+
i = nl + 1
|
|
61
|
+
continue
|
|
62
|
+
yield obj
|
|
63
|
+
i = end
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def iter_records(path: Path, stats: ReadStats | None = None):
|
|
67
|
+
opener = gzip.open if path.suffix == ".gz" else open
|
|
68
|
+
with opener(path, "rt", encoding="utf-8") as fh:
|
|
69
|
+
text = fh.read().strip()
|
|
70
|
+
if not text:
|
|
71
|
+
return
|
|
72
|
+
if text.startswith("["):
|
|
73
|
+
yield from json.loads(text)
|
|
74
|
+
return
|
|
75
|
+
if text.startswith("{"):
|
|
76
|
+
try:
|
|
77
|
+
doc = json.loads(text)
|
|
78
|
+
except json.JSONDecodeError:
|
|
79
|
+
doc = None
|
|
80
|
+
if isinstance(doc, dict):
|
|
81
|
+
if isinstance(doc.get("events"), list):
|
|
82
|
+
for event in doc["events"]:
|
|
83
|
+
try:
|
|
84
|
+
yield json.loads(event.get("message", ""))
|
|
85
|
+
except (json.JSONDecodeError, AttributeError, TypeError):
|
|
86
|
+
if stats:
|
|
87
|
+
stats.bad_lines += 1
|
|
88
|
+
return
|
|
89
|
+
yield doc
|
|
90
|
+
return
|
|
91
|
+
yield from _iter_concatenated(text, stats)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _body(rec: dict):
|
|
95
|
+
body = (rec.get("output") or {}).get("outputBodyJson")
|
|
96
|
+
if isinstance(body, str):
|
|
97
|
+
try:
|
|
98
|
+
body = json.loads(body)
|
|
99
|
+
except json.JSONDecodeError:
|
|
100
|
+
return None
|
|
101
|
+
return body
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _is_invocation_record(rec) -> bool:
|
|
105
|
+
"""Real log records, not offloaded request/response bodies stored next to them in S3."""
|
|
106
|
+
if not isinstance(rec, dict):
|
|
107
|
+
return False
|
|
108
|
+
if rec.get("schemaType") is not None:
|
|
109
|
+
return rec.get("schemaType") == "ModelInvocationLog"
|
|
110
|
+
return "modelId" in rec and ("output" in rec or "operation" in rec)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def aggregate(records, by: str = "model") -> dict[str, Group]:
|
|
114
|
+
"""Group by 'model', 'principal' or 'model+principal'."""
|
|
115
|
+
groups: dict[str, Group] = defaultdict(Group)
|
|
116
|
+
for rec in records:
|
|
117
|
+
if not _is_invocation_record(rec):
|
|
118
|
+
continue
|
|
119
|
+
model = canonical_model_id(rec.get("modelId")) or "?"
|
|
120
|
+
principal = (rec.get("identity") or {}).get("arn") or "?"
|
|
121
|
+
key = {"model": model, "principal": principal}.get(by, f"{model} | {principal}")
|
|
122
|
+
g = groups[key]
|
|
123
|
+
g.calls += 1
|
|
124
|
+
usage = from_response(_body(rec))
|
|
125
|
+
if usage is None:
|
|
126
|
+
g.unparsed += 1
|
|
127
|
+
continue
|
|
128
|
+
g.usage = g.usage + usage
|
|
129
|
+
return dict(groups)
|