callm-toolkit 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. callm_toolkit-0.1.0/.gitignore +21 -0
  2. callm_toolkit-0.1.0/CHANGELOG.md +42 -0
  3. callm_toolkit-0.1.0/LICENSE +21 -0
  4. callm_toolkit-0.1.0/PKG-INFO +363 -0
  5. callm_toolkit-0.1.0/README.md +304 -0
  6. callm_toolkit-0.1.0/examples/async_budgets.py +47 -0
  7. callm_toolkit-0.1.0/examples/demo.tape +21 -0
  8. callm_toolkit-0.1.0/examples/offline_demo.py +167 -0
  9. callm_toolkit-0.1.0/examples/quickstart.py +28 -0
  10. callm_toolkit-0.1.0/examples/structured_output.py +41 -0
  11. callm_toolkit-0.1.0/pyproject.toml +135 -0
  12. callm_toolkit-0.1.0/src/callm/__about__.py +1 -0
  13. callm_toolkit-0.1.0/src/callm/__init__.py +103 -0
  14. callm_toolkit-0.1.0/src/callm/__main__.py +3 -0
  15. callm_toolkit-0.1.0/src/callm/api.py +411 -0
  16. callm_toolkit-0.1.0/src/callm/budgets.py +198 -0
  17. callm_toolkit-0.1.0/src/callm/classify.py +262 -0
  18. callm_toolkit-0.1.0/src/callm/cli.py +349 -0
  19. callm_toolkit-0.1.0/src/callm/config.py +516 -0
  20. callm_toolkit-0.1.0/src/callm/data/__init__.py +1 -0
  21. callm_toolkit-0.1.0/src/callm/data/pricing.json +666 -0
  22. callm_toolkit-0.1.0/src/callm/decorator.py +361 -0
  23. callm_toolkit-0.1.0/src/callm/embeddings.py +135 -0
  24. callm_toolkit-0.1.0/src/callm/errors.py +142 -0
  25. callm_toolkit-0.1.0/src/callm/interception.py +323 -0
  26. callm_toolkit-0.1.0/src/callm/middleware/__init__.py +128 -0
  27. callm_toolkit-0.1.0/src/callm/middleware/cache.py +222 -0
  28. callm_toolkit-0.1.0/src/callm/middleware/cost.py +96 -0
  29. callm_toolkit-0.1.0/src/callm/middleware/fallback.py +75 -0
  30. callm_toolkit-0.1.0/src/callm/middleware/retry.py +52 -0
  31. callm_toolkit-0.1.0/src/callm/middleware/security.py +115 -0
  32. callm_toolkit-0.1.0/src/callm/middleware/telemetry.py +139 -0
  33. callm_toolkit-0.1.0/src/callm/middleware/validator.py +60 -0
  34. callm_toolkit-0.1.0/src/callm/pipeline.py +216 -0
  35. callm_toolkit-0.1.0/src/callm/pricing.py +389 -0
  36. callm_toolkit-0.1.0/src/callm/providers/__init__.py +18 -0
  37. callm_toolkit-0.1.0/src/callm/providers/anthropic.py +256 -0
  38. callm_toolkit-0.1.0/src/callm/providers/base.py +245 -0
  39. callm_toolkit-0.1.0/src/callm/providers/google.py +332 -0
  40. callm_toolkit-0.1.0/src/callm/providers/openai.py +243 -0
  41. callm_toolkit-0.1.0/src/callm/providers/registry.py +125 -0
  42. callm_toolkit-0.1.0/src/callm/py.typed +0 -0
  43. callm_toolkit-0.1.0/src/callm/reports.py +55 -0
  44. callm_toolkit-0.1.0/src/callm/security/__init__.py +20 -0
  45. callm_toolkit-0.1.0/src/callm/security/injection.py +263 -0
  46. callm_toolkit-0.1.0/src/callm/security/pii.py +259 -0
  47. callm_toolkit-0.1.0/src/callm/storage/__init__.py +27 -0
  48. callm_toolkit-0.1.0/src/callm/storage/base.py +193 -0
  49. callm_toolkit-0.1.0/src/callm/storage/memory.py +128 -0
  50. callm_toolkit-0.1.0/src/callm/storage/redis.py +171 -0
  51. callm_toolkit-0.1.0/src/callm/storage/sqlite.py +408 -0
  52. callm_toolkit-0.1.0/src/callm/types.py +376 -0
  53. callm_toolkit-0.1.0/src/callm/validation.py +159 -0
  54. callm_toolkit-0.1.0/tests/conftest.py +149 -0
  55. callm_toolkit-0.1.0/tests/helpers.py +156 -0
  56. callm_toolkit-0.1.0/tests/test_api.py +181 -0
  57. callm_toolkit-0.1.0/tests/test_cache.py +206 -0
  58. callm_toolkit-0.1.0/tests/test_cli.py +193 -0
  59. callm_toolkit-0.1.0/tests/test_core.py +171 -0
  60. callm_toolkit-0.1.0/tests/test_cost.py +277 -0
  61. callm_toolkit-0.1.0/tests/test_decorator.py +304 -0
  62. callm_toolkit-0.1.0/tests/test_extras.py +294 -0
  63. callm_toolkit-0.1.0/tests/test_fallback.py +192 -0
  64. callm_toolkit-0.1.0/tests/test_providers.py +244 -0
  65. callm_toolkit-0.1.0/tests/test_regressions.py +460 -0
  66. callm_toolkit-0.1.0/tests/test_retry.py +253 -0
  67. callm_toolkit-0.1.0/tests/test_security.py +309 -0
  68. callm_toolkit-0.1.0/tests/test_telemetry.py +309 -0
  69. callm_toolkit-0.1.0/tests/test_validation.py +169 -0
@@ -0,0 +1,21 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .venv/
5
+ venv/
6
+ build/
7
+ dist/
8
+ site/
9
+ .coverage
10
+ .coverage.*
11
+ coverage.xml
12
+ htmlcov/
13
+ .pytest_cache/
14
+ .mypy_cache/
15
+ .ruff_cache/
16
+ .DS_Store
17
+ .idea/
18
+ .vscode/
19
+ *.db
20
+ *.db-wal
21
+ *.db-shm
@@ -0,0 +1,42 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0] - 2026-09-15
10
+
11
+ Initial release.
12
+
13
+ ### Added
14
+
15
+ - `@callm` decorator for sync and async functions and methods, intercepting OpenAI
16
+ (`chat.completions.create`), Anthropic (`messages.create`) and Google Gen AI
17
+ (`models.generate_content`) SDK calls and returning the SDK's native response types.
18
+ - `shield` context manager and provider-neutral `complete` / `acomplete`.
19
+ - Retry engine with exponential backoff and jitter that honours `retry-after`,
20
+ `retry-after-ms`, OpenAI and Anthropic rate-limit reset headers and Gemini `RetryInfo`.
21
+ - Response cache: exact-match by default, optional semantic matching (sentence-transformers,
22
+ OpenAI embeddings, or a custom embedder); SQLite, in-memory and Redis backends; TTLs.
23
+ - Provider fallback chains with request translation between OpenAI, Anthropic, Gemini, Ollama
24
+ and OpenAI-compatible providers; provider-specific requests only fall back within their
25
+ provider.
26
+ - Cost tracking from a bundled price table (`callm pricing update` refreshes it), including
27
+ prompt-cache read/write pricing; `max_cost` per call; thread-safe `Budget`s per function,
28
+ session (`callm.budget`) or key (`callm.budget_for`).
29
+ - PII redaction for emails, phone numbers, SSNs, payment cards (Luhn), IPv4 addresses and IBANs
30
+ (checksum), plus optional spaCy person names; mask or block.
31
+ - Heuristic prompt-injection detection (instruction overrides, prompt extraction, role
32
+ hijacking, chat-template tokens, hidden unicode, multilingual variants) with an optional
33
+ Hugging Face classifier; flag or block.
34
+ - Structured output validation for any Pydantic type with automatic re-asking on invalid output.
35
+ - Telemetry records (no prompt text), `on_call` hooks, OpenTelemetry spans, `callm.stats()` and
36
+ `callm.last_call()`.
37
+ - Nested scopes: protections (PII masking, injection detection, the strictest `max_cost`, all
38
+ budgets) from enclosing decorators and shields apply to inner calls.
39
+ - `callm` CLI: `stats`, `calls`, `cache`, `pricing`, `telemetry`, `info`.
40
+
41
+ [Unreleased]: https://github.com/TanbirRamim/callm/compare/v0.1.0...HEAD
42
+ [0.1.0]: https://github.com/TanbirRamim/callm/releases/tag/v0.1.0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Tanbir Hossain Ramim
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,363 @@
1
+ Metadata-Version: 2.5
2
+ Name: callm-toolkit
3
+ Version: 0.1.0
4
+ Summary: The production toolkit for LLM calls: caching, retries, fallback, cost tracking, security and structured output in one decorator.
5
+ Project-URL: Homepage, https://github.com/TanbirRamim/callm
6
+ Project-URL: Documentation, https://tanbirramim.github.io/callm
7
+ Project-URL: Repository, https://github.com/TanbirRamim/callm
8
+ Project-URL: Issues, https://github.com/TanbirRamim/callm/issues
9
+ Project-URL: Changelog, https://github.com/TanbirRamim/callm/blob/main/CHANGELOG.md
10
+ Author: Tanbir Hossain Ramim
11
+ License-Expression: MIT
12
+ License-File: LICENSE
13
+ Keywords: anthropic,caching,fallback,gemini,llm,openai,pii,production,prompt-injection,pydantic,retry
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
25
+ Classifier: Typing :: Typed
26
+ Requires-Python: >=3.10
27
+ Provides-Extra: all
28
+ Requires-Dist: anthropic>=0.40; extra == 'all'
29
+ Requires-Dist: google-genai>=1.0; extra == 'all'
30
+ Requires-Dist: openai>=1.40; extra == 'all'
31
+ Requires-Dist: opentelemetry-api>=1.20; extra == 'all'
32
+ Requires-Dist: pydantic>=2.6; extra == 'all'
33
+ Requires-Dist: redis>=5.0; extra == 'all'
34
+ Requires-Dist: rich>=13.0; extra == 'all'
35
+ Requires-Dist: sentence-transformers>=2.7; extra == 'all'
36
+ Requires-Dist: spacy>=3.7; extra == 'all'
37
+ Requires-Dist: tiktoken>=0.7; extra == 'all'
38
+ Provides-Extra: anthropic
39
+ Requires-Dist: anthropic>=0.40; extra == 'anthropic'
40
+ Provides-Extra: cache
41
+ Requires-Dist: sentence-transformers>=2.7; extra == 'cache'
42
+ Provides-Extra: cli
43
+ Requires-Dist: rich>=13.0; extra == 'cli'
44
+ Provides-Extra: google
45
+ Requires-Dist: google-genai>=1.0; extra == 'google'
46
+ Provides-Extra: openai
47
+ Requires-Dist: openai>=1.40; extra == 'openai'
48
+ Provides-Extra: otel
49
+ Requires-Dist: opentelemetry-api>=1.20; extra == 'otel'
50
+ Provides-Extra: redis
51
+ Requires-Dist: redis>=5.0; extra == 'redis'
52
+ Provides-Extra: security
53
+ Requires-Dist: spacy>=3.7; extra == 'security'
54
+ Provides-Extra: tokens
55
+ Requires-Dist: tiktoken>=0.7; extra == 'tokens'
56
+ Provides-Extra: validation
57
+ Requires-Dist: pydantic>=2.6; extra == 'validation'
58
+ Description-Content-Type: text/markdown
59
+
60
+ <div align="center">
61
+
62
+ <img src="https://raw.githubusercontent.com/TanbirRamim/callm/main/docs/assets/logo.svg" alt="callm" width="96" height="96">
63
+
64
+ # callm
65
+
66
+ **The production toolkit for LLM calls.**
67
+
68
+ Caching, retries, provider fallback, cost tracking, budgets, PII redaction, prompt-injection
69
+ detection, structured output validation and telemetry — in one decorator, on top of the SDKs
70
+ you already use. No proxy. No database server. Zero required dependencies.
71
+
72
+ [![CI](https://github.com/TanbirRamim/callm/actions/workflows/ci.yml/badge.svg)](https://github.com/TanbirRamim/callm/actions/workflows/ci.yml)
73
+ ![Python 3.10+](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12%20%7C%203.13-blue.svg)
74
+ [![PyPI](https://img.shields.io/pypi/v/callm-toolkit.svg)](https://pypi.org/project/callm-toolkit/)
75
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/TanbirRamim/callm/blob/main/LICENSE)
76
+
77
+ [Documentation](https://tanbirramim.github.io/callm) ·
78
+ [Quickstart](#quickstart) ·
79
+ [Features](#features) ·
80
+ [How it works](#how-it-works) ·
81
+ [CLI](#the-callm-cli)
82
+
83
+ </div>
84
+
85
+ ---
86
+
87
+ ```python
88
+ from callm import callm
89
+
90
+ @callm(cache=True, retry=3, fallback=["anthropic/claude-sonnet-5"], max_cost=0.25,
91
+ block_pii=True, detect_injection=True, output_schema=Summary)
92
+ def summarize(text: str) -> Summary:
93
+ return openai.chat.completions.create(
94
+ model="gpt-4o", messages=[{"role": "user", "content": text}]
95
+ )
96
+ ```
97
+
98
+ `summarize()` still calls OpenAI with your code, your client and your API key — but now every
99
+ call is cached, retried on rate limits, failed over to Claude if OpenAI is down, refused if it
100
+ would cost more than 25¢, stripped of emails and phone numbers before it leaves your process,
101
+ scanned for prompt injection, validated into a `Summary` object, and recorded in a local cost
102
+ dashboard.
103
+
104
+ ## Why callm
105
+
106
+ Every team shipping LLM features writes the same production checklist: retry on 429s, cache
107
+ repeated prompts, fall back when a provider has an outage, track spend, keep PII out of prompts,
108
+ catch injection attempts, and make the model return valid JSON. That usually means stitching
109
+ together `tenacity`, a cache, a PII library, an output-parsing library and a pile of glue code —
110
+ or deploying a proxy service.
111
+
112
+ callm is a library: install it, add a decorator, ship.
113
+
114
+ - **No rewrite.** Keep calling `openai`, `anthropic` or `google-genai` directly. callm intercepts
115
+ the SDK call inside decorated functions and hands you back the SDK's own response type.
116
+ - **No infrastructure.** State lives in a local SQLite file (or memory, or Redis if you want a
117
+ shared cache).
118
+ - **No required dependencies.** The core is standard library only; features that need extra
119
+ packages are optional extras.
120
+ - **Sync and async.** Same behaviour for `def` and `async def`, including concurrent tasks.
121
+
122
+ ## Quickstart
123
+
124
+ ```bash
125
+ pip install "callm-toolkit[openai,validation]" # or callm-toolkit[all]
126
+ ```
127
+
128
+ The package is published as `callm-toolkit`; you import it as `callm` and the CLI is `callm`.
129
+
130
+ ```python
131
+ import openai
132
+ from callm import callm
133
+
134
+ client = openai.OpenAI()
135
+
136
+ @callm() # zero-config: retries + cost tracking
137
+ def ask(question: str) -> str:
138
+ response = client.chat.completions.create(
139
+ model="gpt-4o-mini", messages=[{"role": "user", "content": question}]
140
+ )
141
+ return response.choices[0].message.content
142
+
143
+ print(ask("What is the capital of France?"))
144
+ ```
145
+
146
+ ```console
147
+ $ callm stats
148
+ Provider Calls Tokens Cost Cache Hits Saved Errors Latency
149
+ ──────────────────────────────────────────────────────────────────────────
150
+ openai 1 31 $0.0000 0 (0%) $0.00 0 412ms
151
+ ```
152
+
153
+ ### Try it without an API key
154
+
155
+ `examples/offline_demo.py` runs the real OpenAI and Anthropic SDKs against a scripted fake
156
+ server and walks through a rate-limit retry, PII masking, schema validation, a cache hit, a
157
+ fallback to Claude during an outage, a blocked expensive call and a flagged injection:
158
+
159
+ ```bash
160
+ pip install "callm-toolkit[openai,anthropic,validation]"
161
+ export CALLM_HOME=/tmp/callm-demo # keep demo data out of ~/.callm
162
+ python examples/offline_demo.py
163
+ callm stats
164
+ ```
165
+
166
+ ## Features
167
+
168
+ | | Feature | What you get |
169
+ |---|---|---|
170
+ | 🔄 | **Smart retries** | Exponential backoff with full jitter on 408/409/429/5xx/529, timeouts and connection errors. Honours `retry-after`, `retry-after-ms`, OpenAI `x-ratelimit-reset-*`, Anthropic `anthropic-ratelimit-*-reset` and Gemini `RetryInfo`. |
171
+ | 🗄️ | **Response cache** | Exact-match by default; opt-in semantic matching with sentence-transformers, OpenAI embeddings or your own embedder. SQLite, memory or Redis. TTLs. Refusals and invalid output are never cached. |
172
+ | 🔀 | **Provider fallback** | Ordered chains across OpenAI, Anthropic, Gemini, Ollama and any OpenAI-compatible endpoint. Requests are translated between providers, and your code still receives the response type of the SDK it called. |
173
+ | 💰 | **Cost tracking & budgets** | Per-call cost from a bundled price table (refreshable with `callm pricing update`), including prompt-cache read/write pricing. `max_cost` per call, shared `Budget`s per function, session or user — enforced *before* the request is sent. |
174
+ | 🛡️ | **Input security** | PII redaction (emails, phones, SSNs, Luhn-checked cards, IPs, checksum-validated IBANs, optional spaCy names) with stable placeholders. Prompt-injection scoring with a fast heuristic detector and an optional local ML classifier; flag or block. |
175
+ | ✅ | **Structured output** | Pass any Pydantic type as `output_schema`. Invalid output is re-requested with the validation errors appended, and the function returns the validated object. |
176
+ | 📊 | **Telemetry** | Every call records provider, model, tokens, cost, savings, latency, retries, fallbacks and security flags — never prompt text. `callm stats`, `callm calls`, JSON export, `on_call` hooks and OpenTelemetry spans. |
177
+
178
+ ## Examples
179
+
180
+ ### Structured output with automatic repair
181
+
182
+ ```python
183
+ from pydantic import BaseModel
184
+ from callm import callm
185
+
186
+ class Invoice(BaseModel):
187
+ vendor: str
188
+ total: float
189
+ currency: str
190
+
191
+ @callm(output_schema=Invoice, validation_retries=2)
192
+ def extract(text: str):
193
+ return anthropic_client.messages.create(
194
+ model="claude-sonnet-5", max_tokens=1024,
195
+ messages=[{"role": "user", "content": f"Extract the invoice as JSON:\n{text}"}],
196
+ )
197
+
198
+ invoice = extract(raw_text) # -> Invoice(vendor=..., total=..., currency=...)
199
+ ```
200
+
201
+ ### Fallback across providers
202
+
203
+ ```python
204
+ @callm(retry=2, fallback=["anthropic/claude-sonnet-5", "google/gemini-2.5-flash", "ollama/llama3.1"])
205
+ def answer(question: str):
206
+ return openai_client.chat.completions.create(
207
+ model="gpt-4o", messages=[{"role": "user", "content": question}]
208
+ )
209
+ ```
210
+
211
+ If OpenAI keeps returning 429/5xx, callm retries, then sends the same conversation to Claude,
212
+ then Gemini, then a local model — and `answer()` still returns an OpenAI `ChatCompletion`.
213
+ Requests that use provider-specific features (tools, images, response formats) only fall back
214
+ to models of the same provider, so a fallback never silently changes what you asked for.
215
+
216
+ ### Budgets per user
217
+
218
+ ```python
219
+ import callm
220
+
221
+ @callm.callm(max_cost=0.05)
222
+ def chat(messages): ...
223
+
224
+ def handle(user_id: str, messages):
225
+ with callm.budget_for(f"user:{user_id}", limit=2.00):
226
+ return chat(messages) # raises callm.BudgetExceeded once the user spent $2
227
+ ```
228
+
229
+ ### Security without a decorator
230
+
231
+ ```python
232
+ from callm import shield
233
+
234
+ with shield(block_pii=True, detect_injection=True) as s:
235
+ # Any supported SDK call inside the block is protected...
236
+ openai_client.chat.completions.create(model="gpt-4o-mini", messages=user_messages)
237
+ # ...and you can make provider-neutral calls directly.
238
+ response = s.complete(provider="anthropic", model="claude-sonnet-5", messages=user_messages)
239
+ print(response.text, response.cost)
240
+ ```
241
+
242
+ ### Direct, provider-neutral calls
243
+
244
+ ```python
245
+ response = callm.complete("gemini/gemini-2.5-flash", "Summarize: ...", max_tokens=200, cache=True)
246
+ response.text, response.usage.total_tokens, response.cost, response.raw
247
+ ```
248
+
249
+ More in the [cookbook](https://tanbirramim.github.io/callm/cookbook/): a support chatbot, RAG answers with citations and data
250
+ extraction.
251
+
252
+ ## How it works
253
+
254
+ ```text
255
+ your function callm middleware stack
256
+ ───────────── ──────────────────────
257
+ @callm(...) ┌────────────────────────────────────┐
258
+ def summarize(): ─────────► │ 1. Telemetry cost, latency, tokens
259
+ client.chat.completions │ 2. Security injection scan, PII masking
260
+ .create(...) │ 3. Cache exact / semantic lookup
261
+ │ 4. Validator parse + re-ask on invalid output
262
+ │ 5. Fallback next provider when one keeps failing
263
+ │ 6. Cost guard max_cost and budgets, before sending
264
+ │ 7. Retry backoff honouring retry-after
265
+ │ 8. Transport ──► the real SDK call
266
+ └────────────────────────────────────┘
267
+ ```
268
+
269
+ 1. `@callm` sets a scope (a `ContextVar`) while your function runs.
270
+ 2. The official SDK methods (`chat.completions.create`, `messages.create`,
271
+ `models.generate_content`, sync and async) are instrumented on first use. Outside a callm
272
+ scope they call straight through, so importing callm never changes other code.
273
+ 3. Inside a scope, the SDK call is converted into a provider-neutral request and sent through
274
+ the middleware chain. Each layer is independent and only enabled when configured.
275
+ 4. The response is converted back into the SDK's native type before it is returned to you.
276
+
277
+ The middleware is written once as generators and executed by a sync or an async driver, so
278
+ `def` and `async def` functions behave identically.
279
+
280
+ ## Configuration
281
+
282
+ ```python
283
+ import callm
284
+
285
+ callm.configure(
286
+ home="~/.callm", # SQLite database and price overrides
287
+ storage="sqlite", # "sqlite" | "memory" | a storage object
288
+ telemetry=True,
289
+ otel=False, # emit OpenTelemetry spans
290
+ default_retries=2,
291
+ on_call=[print], # called with every CallRecord
292
+ )
293
+ ```
294
+
295
+ | Environment variable | Effect |
296
+ |---|---|
297
+ | `CALLM_HOME` | Data directory (default `~/.callm`) |
298
+ | `CALLM_STORAGE` | `sqlite` or `memory` |
299
+ | `CALLM_TELEMETRY=0` | Do not persist call records |
300
+ | `CALLM_OTEL=1` | Export OpenTelemetry spans |
301
+ | `CALLM_DISABLED=1` | Kill switch: decorated functions run untouched |
302
+
303
+ Every option of `@callm` is documented in the [API reference](https://tanbirramim.github.io/callm/api-reference/).
304
+
305
+ ## The `callm` CLI
306
+
307
+ ```console
308
+ $ callm stats --since 7d # cost, tokens, cache hits and savings by provider
309
+ $ callm stats --by function --json # machine-readable, per function
310
+ $ callm calls --limit 20 # recent calls with retries, fallbacks, flags
311
+ $ callm cache stats | clear
312
+ $ callm pricing show claude-sonnet-5 # USD per 1M tokens
313
+ $ callm pricing update # refresh prices from the LiteLLM price list
314
+ $ callm info # environment and installed extras
315
+ ```
316
+
317
+ ## Installation extras
318
+
319
+ | Extra | Installs | Needed for |
320
+ |---|---|---|
321
+ | `openai` / `anthropic` / `google` | provider SDKs | calling those providers |
322
+ | `validation` | `pydantic>=2` | `output_schema` |
323
+ | `cache` | `sentence-transformers` | semantic caching with local embeddings |
324
+ | `security` | `spacy` | person-name redaction (`PIIConfig(ner=True)`) |
325
+ | `redis` | `redis` | shared cache across hosts |
326
+ | `otel` | `opentelemetry-api` | OpenTelemetry spans |
327
+ | `tokens` | `tiktoken` | exact OpenAI token estimates for the cost guard |
328
+ | `cli` | `rich` | prettier `callm stats` tables |
329
+ | `all` | everything above | |
330
+
331
+ ## Design decisions and limits
332
+
333
+ - **The cache is exact-match unless you opt into semantic matching.** A semantic cache with a
334
+ similarity threshold would happily return the answer for *"Summarize https://a.example"* when
335
+ asked about *"https://b.example"*. Use `cache="semantic"` (or `CacheConfig(semantic=True)`) for
336
+ FAQ-style traffic; semantic matches never cross different system prompts, histories,
337
+ parameters or schemas.
338
+ - **Injection detection flags by default.** Heuristics have false positives, so the default
339
+ logs a warning and records the score; use `InjectionConfig(action="block")` to refuse. No
340
+ detector catches every attack — keep treating model output as untrusted.
341
+ - **Budgets use estimates before a call and actual cost after it.** Set `max_tokens` for a
342
+ tight worst-case estimate; without it the cost guard assumes 1,024 output tokens.
343
+ - **Streaming calls** get security, retries on connection setup, budgets and telemetry, but are
344
+ not cached or validated.
345
+ - **Threads:** the callm scope follows `asyncio` tasks automatically. Work handed to a thread
346
+ pool needs `contextvars.copy_context().run(...)`, or decorate the function running in the
347
+ thread.
348
+ - **Prices** are a bundled snapshot. Run `callm pricing update` or `callm.set_price(...)` for
349
+ current or negotiated rates.
350
+
351
+ ## Contributing
352
+
353
+ Contributions are welcome — see [CONTRIBUTING.md](https://github.com/TanbirRamim/callm/blob/main/CONTRIBUTING.md). The test suite runs
354
+ entirely offline against the real provider SDKs with mocked HTTP transports:
355
+
356
+ ```bash
357
+ uv sync
358
+ uv run pytest
359
+ ```
360
+
361
+ ## License
362
+
363
+ [MIT](https://github.com/TanbirRamim/callm/blob/main/LICENSE)