tonst 0.2.0a1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tonst-0.2.0a1/LICENSE +21 -0
- tonst-0.2.0a1/PKG-INFO +638 -0
- tonst-0.2.0a1/README.md +612 -0
- tonst-0.2.0a1/pyproject.toml +52 -0
- tonst-0.2.0a1/setup.cfg +4 -0
- tonst-0.2.0a1/tonst/__init__.py +81 -0
- tonst-0.2.0a1/tonst/__main__.py +43 -0
- tonst-0.2.0a1/tonst/adapters.py +133 -0
- tonst-0.2.0a1/tonst/cache_structuring.py +405 -0
- tonst-0.2.0a1/tonst/client.py +1089 -0
- tonst-0.2.0a1/tonst/compactor.py +1074 -0
- tonst-0.2.0a1/tonst/gliner_redact.py +322 -0
- tonst-0.2.0a1/tonst/local_model.py +177 -0
- tonst-0.2.0a1/tonst/names.py +253 -0
- tonst-0.2.0a1/tonst/ollama_util.py +57 -0
- tonst-0.2.0a1/tonst/placeholders.py +254 -0
- tonst-0.2.0a1/tonst/providers/__init__.py +37 -0
- tonst-0.2.0a1/tonst/providers/gemini.py +335 -0
- tonst-0.2.0a1/tonst/providers/generic.py +225 -0
- tonst-0.2.0a1/tonst/providers/openai.py +219 -0
- tonst-0.2.0a1/tonst/providers/presets.py +91 -0
- tonst-0.2.0a1/tonst/rag.py +174 -0
- tonst-0.2.0a1/tonst/redact.py +423 -0
- tonst-0.2.0a1/tonst/redact_llm.py +401 -0
- tonst-0.2.0a1/tonst/relevance.py +134 -0
- tonst-0.2.0a1/tonst/savings_log.py +388 -0
- tonst-0.2.0a1/tonst/summarizers.py +221 -0
- tonst-0.2.0a1/tonst/token_count.py +158 -0
- tonst-0.2.0a1/tonst/tool_optimizer.py +451 -0
- tonst-0.2.0a1/tonst/trim.py +75 -0
- tonst-0.2.0a1/tonst.egg-info/PKG-INFO +638 -0
- tonst-0.2.0a1/tonst.egg-info/SOURCES.txt +34 -0
- tonst-0.2.0a1/tonst.egg-info/dependency_links.txt +1 -0
- tonst-0.2.0a1/tonst.egg-info/entry_points.txt +2 -0
- tonst-0.2.0a1/tonst.egg-info/requires.txt +8 -0
- tonst-0.2.0a1/tonst.egg-info/top_level.txt +1 -0
tonst-0.2.0a1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tonst contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
tonst-0.2.0a1/PKG-INFO
ADDED
|
@@ -0,0 +1,638 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tonst
|
|
3
|
+
Version: 0.2.0a1
|
|
4
|
+
Summary: Token optimization and security tool: prompt-caching structuring, PII redaction, and prompt trimming for LLM API calls.
|
|
5
|
+
Author-email: v-nightwolf <info@nightwolf.in>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/v-nightwolf/tonst
|
|
8
|
+
Project-URL: Issues, https://github.com/v-nightwolf/tonst/issues
|
|
9
|
+
Keywords: llm,tokens,cost,prompt-caching,pii,redaction,privacy,anthropic,openai,gemini
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Classifier: Topic :: Security
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: requests>=2.25
|
|
20
|
+
Provides-Extra: local-model
|
|
21
|
+
Provides-Extra: gliner
|
|
22
|
+
Requires-Dist: gliner>=0.2; extra == "gliner"
|
|
23
|
+
Requires-Dist: sentencepiece; extra == "gliner"
|
|
24
|
+
Requires-Dist: protobuf; extra == "gliner"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+

|
|
28
|
+
|
|
29
|
+
# tonst — Token Optimization & Security Tool
|
|
30
|
+
|
|
31
|
+
tonst is a Python library that sits between your application and a paid LLM
|
|
32
|
+
API. Before each request leaves your machine it removes personal data and
|
|
33
|
+
cuts the tokens you pay for; after the response comes back it puts the
|
|
34
|
+
personal data back.
|
|
35
|
+
|
|
36
|
+
It runs in your own process. There is no proxy, server or account, and
|
|
37
|
+
nothing is sent anywhere except the request you were already making.
|
|
38
|
+
|
|
39
|
+
```mermaid
|
|
40
|
+
sequenceDiagram
|
|
41
|
+
participant App as Your app
|
|
42
|
+
participant T as tonst (runs in your process)
|
|
43
|
+
participant API as Your LLM API
|
|
44
|
+
App->>T: prompt or messages (with PII)
|
|
45
|
+
Note over T: 1. Redact PII into placeholders<br/>2. Trim, compact history, filter tools and chunks<br/>3. Order the request for prompt caching
|
|
46
|
+
T->>API: redacted, smaller request
|
|
47
|
+
API-->>T: response (placeholders only)
|
|
48
|
+
Note over T: 4. Put the real values back
|
|
49
|
+
T-->>App: response + savings report
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
**What it does**
|
|
53
|
+
|
|
54
|
+
| Feature | What it saves or protects | Where it helps |
|
|
55
|
+
|---|---|---|
|
|
56
|
+
| PII and secret redaction | Emails, phones, addresses, cards, IPs, API keys and (with GLiNER) names, companies and codenames never reach the provider; answers come back with the real values, and the quality cost is [measured](#privacy-whats-hidden-and-how-answers-hold-up) | Every request |
|
|
57
|
+
| Mechanical trim | Duplicate lines and wasted whitespace | Every request |
|
|
58
|
+
| Tool / MCP definition filtering | Sends only the tool definitions a request needs (−51% to −55% cost in live tests) | Agents and tool-calling apps |
|
|
59
|
+
| Rolling history compaction | Summarizes old turns instead of re-sending or silently dropping them | Long chats and agent loops |
|
|
60
|
+
| RAG context optimization | Drops duplicate and (optionally) irrelevant retrieved chunks | Retrieval pipelines |
|
|
61
|
+
| Prompt-caching structuring | Orders and marks requests so the provider's own cache discounts the repeated part | Stable system prompts and reference docs |
|
|
62
|
+
| Savings log + `tonst stats` | A local record of what was saved, with no prompt content | Monitoring |
|
|
63
|
+
|
|
64
|
+
It works with Anthropic, OpenAI and Gemini, and with any other provider
|
|
65
|
+
through a function you supply. Measured results are summarized
|
|
66
|
+
[below](#results) and detailed in [docs/results.md](docs/results.md).
|
|
67
|
+
|
|
68
|
+
---
|
|
69
|
+
|
|
70
|
+
## Contents
|
|
71
|
+
|
|
72
|
+
- [Install](#install)
|
|
73
|
+
- [Quickstart](#quickstart)
|
|
74
|
+
- [Using tonst in your application](#using-tonst-in-your-application)
|
|
75
|
+
- [1. Connect your model](#1-connect-your-model)
|
|
76
|
+
- [2. Pick the features for your workload](#2-pick-the-features-for-your-workload)
|
|
77
|
+
- [3. Recipes](#3-recipes)
|
|
78
|
+
- [4. Production checklist](#4-production-checklist)
|
|
79
|
+
- [Configuration reference](#configuration-reference)
|
|
80
|
+
- [Results](#results)
|
|
81
|
+
- [Privacy: what's hidden and how answers hold up](#privacy-whats-hidden-and-how-answers-hold-up)
|
|
82
|
+
- [Limitations](#limitations)
|
|
83
|
+
- [Documentation](#documentation)
|
|
84
|
+
- [Whitepaper, license and contributing](#whitepaper-license-and-contributing)
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## Install
|
|
89
|
+
|
|
90
|
+
Requires Python 3.10+. tonst isn't on PyPI yet; install it from GitHub:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
pip install "git+https://github.com/v-nightwolf/tonst.git"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
For production, pin a specific commit or tag so an update never surprises you:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
pip install "git+https://github.com/v-nightwolf/tonst.git@<commit-or-tag>"
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
The only required dependency is `requests`. Two optional pieces add
|
|
103
|
+
capability:
|
|
104
|
+
|
|
105
|
+
| Optional piece | What it adds | Install |
|
|
106
|
+
|---|---|---|
|
|
107
|
+
| GLiNER | Free-text PII detection (names, companies, codenames) on CPU, ~0.2–1 s per request | `pip install "tonst[gliner] @ git+https://github.com/v-nightwolf/tonst.git"` (pulls in `torch`, `transformers`, `sentencepiece`, `protobuf`) |
|
|
108
|
+
| [Ollama](https://ollama.com) | A local model for history summaries or compression, e.g. `ollama pull gemma2:2b` | Separate app, not a pip package |
|
|
109
|
+
|
|
110
|
+
Neither is needed for the quickstart. If an optional piece is missing, the
|
|
111
|
+
feature that uses it is skipped rather than breaking the request. If GLiNER is
|
|
112
|
+
selected but can't load, tonst logs a loud warning that names are **not**
|
|
113
|
+
being hidden.
|
|
114
|
+
|
|
115
|
+
## Quickstart
|
|
116
|
+
|
|
117
|
+
Wrap the function you already use to call your model. tonst takes a
|
|
118
|
+
function from prompt text to response text:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
import anthropic
|
|
122
|
+
from tonst import TonstClient
|
|
123
|
+
|
|
124
|
+
api = anthropic.Anthropic() # reads ANTHROPIC_API_KEY
|
|
125
|
+
|
|
126
|
+
def call_model(prompt: str) -> str:
|
|
127
|
+
resp = api.messages.create(
|
|
128
|
+
model="claude-sonnet-4-6", max_tokens=1024,
|
|
129
|
+
messages=[{"role": "user", "content": prompt}],
|
|
130
|
+
)
|
|
131
|
+
return "".join(b.text for b in resp.content if b.type == "text")
|
|
132
|
+
|
|
133
|
+
client = TonstClient(call_fn=call_model)
|
|
134
|
+
|
|
135
|
+
answer, report = client.query(
|
|
136
|
+
"Customer jo@example.com says order 4471 arrived broken.\n"
|
|
137
|
+
"Customer jo@example.com says order 4471 arrived broken.\n"
|
|
138
|
+
"Draft a short, polite reply."
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
print(answer) # real email address restored in the reply
|
|
142
|
+
print(report.redacted_fields) # 1 -- the model saw [[EMAIL_…]], not the address
|
|
143
|
+
print(report.tokens_saved, report.percent_saved) # the duplicate line was trimmed
|
|
144
|
+
print(report.local_overhead_ms) # tonst's own time before the API call
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Any provider works the same way: `call_model` is your code, so it can call
|
|
148
|
+
OpenAI, Gemini, Bedrock, a self-hosted model or an internal gateway.
|
|
149
|
+
|
|
150
|
+
---
|
|
151
|
+
|
|
152
|
+
## Using tonst in your application
|
|
153
|
+
|
|
154
|
+
### 1. Connect your model
|
|
155
|
+
|
|
156
|
+
tonst never calls a provider on its own. You give it one of two functions:
|
|
157
|
+
|
|
158
|
+
| | `call_fn(prompt: str) -> str` | `messages_fn(messages: list[dict]) -> str` |
|
|
159
|
+
|---|---|---|
|
|
160
|
+
| Receives | One flattened, redacted prompt string | The redacted `{"role", "content"}` list, system messages included |
|
|
161
|
+
| Best for | Single-shot requests: classification, extraction, drafting | Chat apps and anything that needs roles or prompt caching |
|
|
162
|
+
| Can report real usage | No | Yes: return `(text, usage)` |
|
|
163
|
+
|
|
164
|
+
With `messages_fn`, use the adapters in `tonst.adapters` to convert the
|
|
165
|
+
message list for your provider and to read the provider's usage block
|
|
166
|
+
back. Returning usage lets tonst report real billed and cached tokens, and
|
|
167
|
+
lets cache-aware compaction see how well the provider is actually caching.
|
|
168
|
+
|
|
169
|
+
**Anthropic**, with prompt-caching breakpoints set for you:
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
import anthropic
|
|
173
|
+
from tonst import TonstClient
|
|
174
|
+
from tonst.adapters import to_anthropic, usage_from_anthropic
|
|
175
|
+
|
|
176
|
+
api = anthropic.Anthropic()
|
|
177
|
+
|
|
178
|
+
def call_claude(messages):
|
|
179
|
+
resp = api.messages.create(model="claude-sonnet-4-6", max_tokens=1024,
|
|
180
|
+
**to_anthropic(messages))
|
|
181
|
+
text = "".join(b.text for b in resp.content if b.type == "text")
|
|
182
|
+
return text, usage_from_anthropic(resp)
|
|
183
|
+
|
|
184
|
+
client = TonstClient(messages_fn=call_claude)
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
**OpenAI**, which caches repeated prefixes automatically:
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
from openai import OpenAI
|
|
191
|
+
from tonst.adapters import to_openai, usage_from_openai
|
|
192
|
+
|
|
193
|
+
oa = OpenAI()
|
|
194
|
+
|
|
195
|
+
def call_openai(messages):
|
|
196
|
+
resp = oa.chat.completions.create(model="gpt-4o", messages=to_openai(messages))
|
|
197
|
+
return resp.choices[0].message.content, usage_from_openai(resp)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
**Gemini** (REST), where implicit caching is automatic but best-effort:
|
|
201
|
+
|
|
202
|
+
```python
|
|
203
|
+
import os, requests
|
|
204
|
+
from tonst.adapters import to_gemini, usage_from_gemini
|
|
205
|
+
|
|
206
|
+
URL = "https://generativelanguage.googleapis.com/v1beta/models/gemini-3.8-flash:generateContent"
|
|
207
|
+
|
|
208
|
+
def call_gemini(messages):
|
|
209
|
+
resp = requests.post(URL, headers={"x-goog-api-key": os.environ["GEMINI_API_KEY"]},
|
|
210
|
+
json=to_gemini(messages), timeout=120).json()
|
|
211
|
+
parts = resp["candidates"][0]["content"]["parts"]
|
|
212
|
+
return "".join(p.get("text", "") for p in parts if not p.get("thought")), usage_from_gemini(resp)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
A client built with `messages_fn` also handles `query()`, `query_structured()`
|
|
216
|
+
and `query_rag()`: they send it a single user message.
|
|
217
|
+
|
|
218
|
+
### 2. Pick the features for your workload
|
|
219
|
+
|
|
220
|
+
Everything except regex redaction and mechanical trim is off until you
|
|
221
|
+
turn it on.
|
|
222
|
+
|
|
223
|
+
| Your workload | Turn on | Why |
|
|
224
|
+
|---|---|---|
|
|
225
|
+
| Any app sending user text | `redaction_backend="regex"` (default); `"gliner"` if prompts contain names or free-text personal data | Keeps PII off the provider |
|
|
226
|
+
| Chat app or support bot | `query_messages()` with a `RollingSummary` per conversation, `background_summary=True` | Bounded history without losing facts, no added latency |
|
|
227
|
+
| Chat app on a provider with prompt caching | Also `compaction_cache_aware=True` and pass usage back from `messages_fn` | Summarizes only when it pays off against the cache |
|
|
228
|
+
| Agent or tool-calling app | `select_tools()` / `ToolSession` on your tool list | Tool definitions are often most of the prompt |
|
|
229
|
+
| Huge tool catalog on Anthropic | `build_anthropic_deferred_tools()` | Uses Anthropic's own tool search |
|
|
230
|
+
| RAG pipeline | `query_rag()` | Drops duplicate and irrelevant chunks |
|
|
231
|
+
| Large stable system prompt or reference docs | `query_structured()`, or `messages_fn` with `to_anthropic()` | Puts the repeated part first so it caches |
|
|
232
|
+
| You want to see the savings | `savings_log=True`, then `tonst stats` | Local log, no prompt content |
|
|
233
|
+
|
|
234
|
+
### 3. Recipes
|
|
235
|
+
|
|
236
|
+
#### Chat app or support bot
|
|
237
|
+
|
|
238
|
+
Keep one `RollingSummary` per conversation and pass the full history each
|
|
239
|
+
turn. tonst sends the recent turns verbatim, folds older turns into a
|
|
240
|
+
structured summary (goal, decisions, key facts, open items), and carries
|
|
241
|
+
exact references such as order numbers and case IDs forward word for word.
|
|
242
|
+
|
|
243
|
+
```python
|
|
244
|
+
from tonst import TonstClient, RollingSummary, AnthropicSummarizer
|
|
245
|
+
|
|
246
|
+
client = TonstClient(
|
|
247
|
+
messages_fn=call_claude, # from step 1
|
|
248
|
+
use_history_compaction=True,
|
|
249
|
+
compaction_summarizer=AnthropicSummarizer(), # Claude Haiku 4.5; omit to use local Ollama
|
|
250
|
+
compaction_cache_aware=True, # hold summaries back while the cache is cheaper
|
|
251
|
+
compaction_token_threshold=3000, # summarize in batches of ~3k tokens
|
|
252
|
+
savings_log=True,
|
|
253
|
+
app_name="support-bot",
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
state = RollingSummary() # one per conversation
|
|
257
|
+
history = [{"role": "system", "content": "You are Acme's support assistant."}]
|
|
258
|
+
|
|
259
|
+
def handle_turn(user_text: str) -> str:
|
|
260
|
+
history.append({"role": "user", "content": user_text})
|
|
261
|
+
reply, report = client.query_messages(history, keep_last_n=6,
|
|
262
|
+
rolling_state=state, background_summary=True)
|
|
263
|
+
history.append({"role": "assistant", "content": reply})
|
|
264
|
+
return reply
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
To keep a conversation across requests (a web app, a queue worker), store
|
|
268
|
+
the state next to the conversation:
|
|
269
|
+
|
|
270
|
+
```python
|
|
271
|
+
client.wait_for_background_work(timeout=30) # let a running summary finish first
|
|
272
|
+
db.save(conversation_id, state.to_dict()) # plain JSON: summary text, counters, pinned references
|
|
273
|
+
# next request:
|
|
274
|
+
state = RollingSummary.from_dict(db.load(conversation_id))
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
Choosing the summarizer:
|
|
278
|
+
|
|
279
|
+
| Summarizer | Cost | Quality in live tests | Needs |
|
|
280
|
+
|---|---|---|---|
|
|
281
|
+
| `AnthropicSummarizer()` (Claude Haiku 4.5) | ~1 cent over a 24-turn chat | 8/8 facts kept | `ANTHROPIC_API_KEY` |
|
|
282
|
+
| `GeminiSummarizer()` (Gemini 3.5 Flash-Lite) | ~0.3 cents over a 20-turn chat | 8/8 facts kept | `GEMINI_API_KEY` |
|
|
283
|
+
| Local Ollama model (default, `gemma2:2b`) | Free | 5/8 facts kept; slower | Ollama running |
|
|
284
|
+
|
|
285
|
+
Summarizers only ever receive already-redacted text. If one fails, the
|
|
286
|
+
turns stay verbatim and tonst retries before falling back to dropping
|
|
287
|
+
them; `report.history_tokens_lost` shows anything that was dropped.
|
|
288
|
+
|
|
289
|
+
#### Agent or tool-calling app
|
|
290
|
+
|
|
291
|
+
tonst doesn't run your agent loop, so use the tool filter directly where
|
|
292
|
+
you build each request. `ToolSession` only ever adds tools during a
|
|
293
|
+
conversation, never removes them, so the tool list stays byte-identical
|
|
294
|
+
between turns and keeps hitting the provider's cache.
|
|
295
|
+
|
|
296
|
+
```python
|
|
297
|
+
from tonst import ToolSession, select_tools
|
|
298
|
+
|
|
299
|
+
session = ToolSession(ALL_TOOLS, top_k=8) # one per conversation or agent run
|
|
300
|
+
|
|
301
|
+
def next_request(messages, user_text):
|
|
302
|
+
sel = session.select(user_text)
|
|
303
|
+
# sel.tools: the definitions to send (original objects, original order)
|
|
304
|
+
# sel.fell_back: True when tonst wasn't confident and kept every tool
|
|
305
|
+
return api.messages.create(model="claude-sonnet-4-6", max_tokens=1024,
|
|
306
|
+
tools=sel.tools, messages=messages)
|
|
307
|
+
|
|
308
|
+
# Stateless, single request:
|
|
309
|
+
sel = select_tools(ALL_TOOLS, "Create a Jira ticket for the refund bug", top_k=5)
|
|
310
|
+
```
|
|
311
|
+
|
|
312
|
+
`select_tools()` accepts Anthropic, OpenAI (Chat Completions and Responses)
|
|
313
|
+
and Gemini function-declaration shapes as they are. It ranks by keyword
|
|
314
|
+
relevance, and when a request shares fewer than two words with every tool
|
|
315
|
+
it sends all of them instead of guessing. For catalogs of hundreds of
|
|
316
|
+
tools on Anthropic, `build_anthropic_deferred_tools(ALL_TOOLS)` hands the
|
|
317
|
+
choice to Anthropic's tool search instead. See
|
|
318
|
+
[docs/tools-and-rag.md](docs/tools-and-rag.md).
|
|
319
|
+
|
|
320
|
+
#### RAG pipeline
|
|
321
|
+
|
|
322
|
+
```python
|
|
323
|
+
answer, report = client.query_rag(
|
|
324
|
+
question="How long do refunds take?",
|
|
325
|
+
chunks=retrieved_chunks, # strings, or dicts with "text"/"content"
|
|
326
|
+
system="Answer only from the provided context.",
|
|
327
|
+
top_k=4, # optional: also drop low-relevance chunks
|
|
328
|
+
)
|
|
329
|
+
print(report.chunks_in, "→", report.chunks_sent)
|
|
330
|
+
```
|
|
331
|
+
|
|
332
|
+
Duplicates and near-duplicates are always removed. Relevance filtering
|
|
333
|
+
happens only if you ask for it (`top_k`, `min_relative_score` or
|
|
334
|
+
`max_tokens`), and is skipped when no chunk clearly matches the question.
|
|
335
|
+
|
|
336
|
+
#### Large stable prompts and prompt caching
|
|
337
|
+
|
|
338
|
+
Provider caches only discount a repeated prefix. Put the part that never
|
|
339
|
+
changes first and the per-request part last:
|
|
340
|
+
|
|
341
|
+
```python
|
|
342
|
+
from tonst import PromptParts
|
|
343
|
+
|
|
344
|
+
parts = PromptParts(
|
|
345
|
+
system="You are a contracts analyst.",
|
|
346
|
+
stable_blocks=[playbook_text, clause_library], # identical on every call
|
|
347
|
+
variable=user_question, # changes every call
|
|
348
|
+
)
|
|
349
|
+
answer, report = client.query_structured(parts)
|
|
350
|
+
```
|
|
351
|
+
|
|
352
|
+
Redaction placeholders are deterministic hashes, so a stable block that
|
|
353
|
+
contains PII redacts to the same bytes every time and still caches. For
|
|
354
|
+
Anthropic's explicit `cache_control` markers, either use `messages_fn`
|
|
355
|
+
with `to_anthropic()`, or build the request yourself with
|
|
356
|
+
`redact_and_trim_parts()` + `build_anthropic_cache_request()`. Provider
|
|
357
|
+
details, minimum cacheable lengths and the generic config for other
|
|
358
|
+
providers are in [docs/caching-and-providers.md](docs/caching-and-providers.md).
|
|
359
|
+
|
|
360
|
+
### 4. Production checklist
|
|
361
|
+
|
|
362
|
+
**Privacy**
|
|
363
|
+
- The model provider receives only redacted text. The placeholder → value
|
|
364
|
+
mapping stays in memory for the duration of the call and is never logged.
|
|
365
|
+
- Optional remote helpers (`AnthropicSummarizer`, `GeminiSummarizer`,
|
|
366
|
+
`AnthropicTokenCounter`, `GeminiTokenCounter`) also receive only
|
|
367
|
+
redacted text.
|
|
368
|
+
- The savings log stores counts and redaction *labels* (e.g. `EMAIL: 2`),
|
|
369
|
+
never prompt text, values or placeholder hashes.
|
|
370
|
+
- `regex` catches structured PII only. If your prompts contain names or
|
|
371
|
+
free-text personal details, use `redaction_backend="gliner"`, and test
|
|
372
|
+
recall on your own data first ([docs/redaction.md](docs/redaction.md)).
|
|
373
|
+
|
|
374
|
+
**Latency**
|
|
375
|
+
|
|
376
|
+
| Step | Typical added time |
|
|
377
|
+
|---|---|
|
|
378
|
+
| Regex redaction, trim, tool selection, compaction bookkeeping | a few milliseconds |
|
|
379
|
+
| GLiNER redaction (CPU) | ~150 ms – 1.3 s, depending on hardware |
|
|
380
|
+
| Background summary | 0 on the request path |
|
|
381
|
+
| Blocking summary with a local 2B model | 4–12 s on the turn it runs |
|
|
382
|
+
| Exact token counter | one extra network round trip per count (~300 ms measured on Gemini) |
|
|
383
|
+
|
|
384
|
+
Every report carries per-step timings (`redaction_ms`, `compaction_ms`,
|
|
385
|
+
`call_ms`, `local_overhead_ms`, …) so you can check this on your own
|
|
386
|
+
hardware.
|
|
387
|
+
|
|
388
|
+
**State and concurrency**
|
|
389
|
+
- `RollingSummary` and `ToolSession` are per-conversation state; persist
|
|
390
|
+
them with `to_dict()` (`ToolSession.load_state()` / `RollingSummary.from_dict()`
|
|
391
|
+
to restore). They hold only redacted text.
|
|
392
|
+
- Each `RollingSummary` has its own lock. Background summaries run on one
|
|
393
|
+
worker thread per client; call `client.wait_for_background_work()`
|
|
394
|
+
before shutting down or saving state.
|
|
395
|
+
- Concurrency has been tested for the GLiNER redaction path (4 workers:
|
|
396
|
+
same results, higher latency). Sharing one `TonstClient` across many
|
|
397
|
+
threads hasn't been load-tested; one client per worker process is the
|
|
398
|
+
conservative choice.
|
|
399
|
+
|
|
400
|
+
**Failure behaviour**
|
|
401
|
+
- Every optional step fails soft: if Ollama, GLiNER or a remote summarizer
|
|
402
|
+
is unavailable, that step is skipped and the request still goes out.
|
|
403
|
+
- The one lossy fallback is history compaction: if summaries keep failing,
|
|
404
|
+
old turns are eventually dropped. `report.history_tokens_lost` and
|
|
405
|
+
`tonst stats` show when that happens.
|
|
406
|
+
- The API call itself is your code. tonst doesn't retry it or change its
|
|
407
|
+
timeouts.
|
|
408
|
+
|
|
409
|
+
**Monitoring**
|
|
410
|
+
|
|
411
|
+
```bash
|
|
412
|
+
tonst stats # all apps, from ~/.tonst/savings.jsonl
|
|
413
|
+
tonst stats --app support-bot --since 2026-09-01
|
|
414
|
+
tonst stats --json # for dashboards
|
|
415
|
+
```
|
|
416
|
+
|
|
417
|
+
Set `TONST_SAVINGS_LOG` to put the log somewhere else, and
|
|
418
|
+
`input_price_per_million=` on the client to see estimated dollars. Token
|
|
419
|
+
counts are chars/4 estimates unless you pass `token_counter=` (see
|
|
420
|
+
[docs/measurement.md](docs/measurement.md)); `messages_fn` usage adds
|
|
421
|
+
the provider's real prompt and cached-token counts to each entry.
|
|
422
|
+
|
|
423
|
+
---
|
|
424
|
+
|
|
425
|
+
## Configuration reference
|
|
426
|
+
|
|
427
|
+
The `TonstClient` options you're most likely to set. The full list, with
|
|
428
|
+
reasoning for each default, is in the `TonstClient.__init__` docstring
|
|
429
|
+
(`tonst/client.py`).
|
|
430
|
+
|
|
431
|
+
| Option | Default | Meaning |
|
|
432
|
+
|---|---|---|
|
|
433
|
+
| `call_fn` / `messages_fn` | — | How tonst calls your model; pass at least one ([step 1](#1-connect-your-model)) |
|
|
434
|
+
| `redaction_backend` | `"regex"` | `"none"`, `"regex"`, `"gliner"` or `"ollama"` |
|
|
435
|
+
| `use_history_compaction` | `False` | Summarize old turns in `query_messages()` instead of only dropping them |
|
|
436
|
+
| `compaction_summarizer` | local Ollama | `AnthropicSummarizer()`, `GeminiSummarizer()` or any `fn(prompt, model, timeout) -> str` |
|
|
437
|
+
| `compaction_token_threshold` | `3000` | How many tokens of old turns to batch into one summary |
|
|
438
|
+
| `compaction_cache_aware` | `False` | Postpone summaries that wouldn't pay off against the provider's cache |
|
|
439
|
+
| `compaction_cache_pricing` | `"anthropic"` | `"anthropic"`, `"gemini"` or a `(write, read)` price-multiplier tuple |
|
|
440
|
+
| `use_local_compression` | `False` | Rewrite prompts shorter with a local Ollama model (single-shot requests only) |
|
|
441
|
+
| `local_model` | `"gemma2:2b"` | Ollama model for local steps |
|
|
442
|
+
| `savings_log` | `None` | `True`, a file path, or a `SavingsLog` |
|
|
443
|
+
| `app_name`, `input_price_per_million` | `None` | Labels and pricing for the savings log |
|
|
444
|
+
| `token_counter` | `None` | `AnthropicTokenCounter(...)` / `GeminiTokenCounter(...)` for exact counts |
|
|
445
|
+
| `placeholder_style` | `"hash"` | `"hash"` (stable, keeps prompt caching working) or `"readable"` (`[[NAME_1]]`) |
|
|
446
|
+
| `placeholder_hint` | `True` | Tell the model what the placeholders are (strongly recommended for Claude) |
|
|
447
|
+
| `email_style` | `"split"` | `[[EMAIL_1]]@[[DOMAIN_1]]`; `"whole"` for one token per address |
|
|
448
|
+
| `extra_redaction` | `()` | Add `"ACCOUNT_ID"` and/or `"MONEY"` (money breaks arithmetic tasks) |
|
|
449
|
+
| `restore_secrets` | `False` | Put withheld keys back into answers (off: shown as `[REDACTED]`) |
|
|
450
|
+
| `secret_notice` | `False` | Append a one-line "secrets were withheld" note to the answer |
|
|
451
|
+
|
|
452
|
+
| Method | Use it for |
|
|
453
|
+
|---|---|
|
|
454
|
+
| `query(prompt)` | One flat prompt |
|
|
455
|
+
| `query_messages(messages, keep_last_n=6, rolling_state=None, background_summary=False)` | Chat history |
|
|
456
|
+
| `query_structured(PromptParts(...))` | Stable prefix + variable question |
|
|
457
|
+
| `query_rag(question, chunks, ...)` | Retrieved context |
|
|
458
|
+
|
|
459
|
+
Each returns `(response_text, OptimizationReport)`. Useful report fields:
|
|
460
|
+
`original_tokens`, `sent_tokens`, `tokens_saved`, `percent_saved`,
|
|
461
|
+
`redacted_fields`, `redacted_types`, `secrets_withheld`, `history_*`, `chunks_in` /
|
|
462
|
+
`chunks_sent`, `provider_prompt_tokens` / `provider_cached_tokens`, and
|
|
463
|
+
the `*_ms` timings.
|
|
464
|
+
|
|
465
|
+
---
|
|
466
|
+
|
|
467
|
+
## Results
|
|
468
|
+
|
|
469
|
+
Headline numbers. Each one says how it was measured; the full tables,
|
|
470
|
+
methods and every intermediate run are in [docs/results.md](docs/results.md).
|
|
471
|
+
|
|
472
|
+
| What | Result | How it was measured |
|
|
473
|
+
|---|---|---|
|
|
474
|
+
| Tool filtering, 36 tools, 30 tasks | Cost −51% (Claude Sonnet 4.6) and −55% (Gemini 3.8 Flash), with the same success rate as sending every tool | Live API |
|
|
475
|
+
| Rolling compaction, 20-turn chat on Gemini 3.8 Flash | Cost −15.1%, prompt tokens −43%, 8/8 facts kept, p95 latency 4.0 s → 2.7 s | Live API |
|
|
476
|
+
| Rolling compaction, 24-turn chat on Claude Sonnet 4.6 | Cost −2.7% with Haiku summaries (8/8 facts); caching already makes old turns cheap there | Live API |
|
|
477
|
+
| Cache-aware compaction, short chats | Avoids a +14.9% loss that summarizing too early caused in a 12-turn Claude chat | Live API |
|
|
478
|
+
| Prompt caching, Gemini 3.1 Flash-Lite, 6 domains | Net cost −53% across 24 calls | Live API |
|
|
479
|
+
| Redaction + trim + compression, 360 prompts in 6 domains | Tokens −21.1%; 100% structured-PII recall with 0 leaks; 87.8% free-text PII recall with GLiNER | Local pipeline (API mocked) |
|
|
480
|
+
| Mechanical trim on long, redundant inputs (privacy benchmark's heavy cases) | Input tokens −39% to −49% on re-quoted email threads, −40% to −48% on padded meeting transcripts; no saving on log dumps or small RAG sets | Live API |
|
|
481
|
+
| Answer quality with redaction on, 100 work prompts | −0.46 (Claude Sonnet 4.6) and −0.42 (Gemini 3.8 Flash) on a 1–10 judge vs. unredacted; 100% of must-have values kept (preliminary) | Live API, blind LLM judge |
|
|
482
|
+
| Compaction on long chats with caching | −18% at 40 turns, −55% at 100 turns | Simulation, calibrated to the live runs |
|
|
483
|
+
|
|
484
|
+
Savings depend on your traffic. A short, clean prompt gets 0% from
|
|
485
|
+
trimming, and tonst reports 0% rather than inventing a saving.
|
|
486
|
+
|
|
487
|
+
---
|
|
488
|
+
|
|
489
|
+
## Privacy: what's hidden and how answers hold up
|
|
490
|
+
|
|
491
|
+
Redaction is only useful if the answers stay good. tonst measures that with a
|
|
492
|
+
published benchmark, and hides more than the obvious fields.
|
|
493
|
+
|
|
494
|
+
### Does hiding the data make answers worse?
|
|
495
|
+
|
|
496
|
+
A little, and we measured how much. The benchmark in
|
|
497
|
+
[`experiments/privacy_quality/`](experiments/privacy_quality/) sends 120
|
|
498
|
+
realistic work prompts — support replies, contracts, HR notes, invoices,
|
|
499
|
+
config files with keys, meeting notes, small tables, translations, long
|
|
500
|
+
email threads, and 10 hold-out prompts written after the detectors were
|
|
501
|
+
tuned — to Claude Sonnet 4.6 and Gemini 3.8 Flash, once as-is and once
|
|
502
|
+
through tonst. A separate model grades each pair blind, in random
|
|
503
|
+
order. All people, companies and keys in the prompts are invented.
|
|
504
|
+
|
|
505
|
+
> **Preliminary numbers.** These come from development runs on
|
|
506
|
+
> 2026-09-28. A single final run on the release code will replace them
|
|
507
|
+
> before publishing (`python experiments/privacy_quality/run.py`).
|
|
508
|
+
|
|
509
|
+
| | Claude Sonnet 4.6 | Gemini 3.8 Flash |
|
|
510
|
+
|---|---|---|
|
|
511
|
+
| Answer quality, masked vs. as-is (1–10 judge, 100 main prompts) | −0.46 | −0.42 |
|
|
512
|
+
| Answers containing every must-have value (names, totals, IDs) | 100% | 100% |
|
|
513
|
+
| Placeholders left in answers | 0 | 0 |
|
|
514
|
+
|
|
515
|
+
Most of the remaining gap has a known cause: the model can't write a name
|
|
516
|
+
it never saw in another script (e.g. Hindi), can't tell someone's gender
|
|
517
|
+
from a placeholder, and occasionally mixes up whose contact details are
|
|
518
|
+
whose. Hiding money amounts breaks arithmetic, which is why that category
|
|
519
|
+
is off by default.
|
|
520
|
+
|
|
521
|
+
### What gets hidden
|
|
522
|
+
|
|
523
|
+
| Category | Examples | How | Default |
|
|
524
|
+
|---|---|---|---|
|
|
525
|
+
| Emails | `priya.nair@veltrix.io` | pattern | on |
|
|
526
|
+
| Phone numbers | `+91 98450 21733`, `(415) 555-0144`, `090000 12345` | pattern | on |
|
|
527
|
+
| Postal addresses | `1180 Folsom Street, San Francisco, CA 94103`, `Flat 9B, …, Gurugram 122003` | pattern | on |
|
|
528
|
+
| Secrets | Anthropic/OpenAI/AWS/GitHub/Slack/Google keys, JWTs, private keys, `password=…` | pattern | on |
|
|
529
|
+
| Cards, SSN-like IDs, IP addresses | `4111 1111 1111 1111`, `10.24.8.117` | pattern | on |
|
|
530
|
+
| Names, companies, project codenames | `Omar Haddad`, `Veltrix Logistics`, `Project Bluefin` | [GLiNER](https://github.com/urchade/GLiNER), a small local model, plus name rules | with `redaction_backend="gliner"` |
|
|
531
|
+
| Account / invoice / order numbers | `INV-22243`, `customer #928381` | pattern | opt-in: `extra_redaction=["ACCOUNT_ID"]` |
|
|
532
|
+
| Money amounts | `$14,821.32`, `₹12 lakh` | pattern | opt-in: `extra_redaction=["MONEY"]` |
|
|
533
|
+
|
|
534
|
+
In the benchmark (default settings plus GLiNER), no names, emails,
|
|
535
|
+
addresses, secrets, codenames or IP addresses reached either provider;
|
|
536
|
+
one company mention ("the Brightwell Health clinic") and one phone format
|
|
537
|
+
(since fixed) did. Detection is never perfect: test on your own data
|
|
538
|
+
before relying on it. Public services such as `api.anthropic.com` are
|
|
539
|
+
deliberately left visible — hiding them from the provider protects nothing.
|
|
540
|
+
|
|
541
|
+
### How the placeholders work
|
|
542
|
+
|
|
543
|
+
- **Keyed, stable placeholders (default).** `[[EMAIL_8ddc0a70]]` is an HMAC
|
|
544
|
+
of the value with a key that never leaves your machine
|
|
545
|
+
(`~/.tonst/placeholder.key`, or `TONST_PLACEHOLDER_KEY`). The same
|
|
546
|
+
value gets the same placeholder on every call, so provider prompt caching
|
|
547
|
+
keeps working, and a provider can't confirm a guessed value by hashing it.
|
|
548
|
+
`placeholder_style="readable"` gives `[[EMAIL_1]]`-style numbering instead.
|
|
549
|
+
- **One placeholder per person.** "Omar Haddad" → `[[NAME_1]]`, a later
|
|
550
|
+
"Omar" → `[[NAME_1.first]]`, "Dr. Haddad" → `Dr. [[NAME_1.last]]`, so the
|
|
551
|
+
model can write "Hi [[NAME_1.first]]" and the reply says "Hi Omar".
|
|
552
|
+
- **Emails keep their shape.** `[[EMAIL_1]]@[[DOMAIN_1]]`: addresses at the
|
|
553
|
+
same company share a domain placeholder, so "group these by company"
|
|
554
|
+
still works.
|
|
555
|
+
- **A short note to the model** (on by default, only when something was
|
|
556
|
+
hidden) says the tokens stand for real values. Extra lines are added only
|
|
557
|
+
when they apply: first/last-name parts when a person was hidden, whose
|
|
558
|
+
contact details are whose, and that `[[SECRET_…]]` tokens are exposed
|
|
559
|
+
credentials. About 45 tokens, up to ~120 with every line; in chats the
|
|
560
|
+
fixed part sits in the system message, where prompt caching makes repeat
|
|
561
|
+
reads ~90% cheaper. Without the note, Claude treated placeholders as
|
|
562
|
+
template blanks in ~40% of answers.
|
|
563
|
+
- **Secrets never come back.** Keys are withheld from the provider and shown
|
|
564
|
+
as `[REDACTED]` in answers; `report.secrets_withheld` counts them.
|
|
565
|
+
- **Tolerant restore.** Placeholders the model reformats (`NAME_1`,
|
|
566
|
+
`[NAME_1]`) are still restored; ones it invents become neutral blanks
|
|
567
|
+
(`[email]`) instead of raw tokens. `StreamRestorer` does the same for
|
|
568
|
+
streamed answers.
|
|
569
|
+
|
|
570
|
+
---
|
|
571
|
+
|
|
572
|
+
## Limitations
|
|
573
|
+
|
|
574
|
+
- **Token counts are estimates by default** (characters ÷ 4). That was
|
|
575
|
+
within 2% of Gemini's real counts but 1.77× too low for Claude requests
|
|
576
|
+
with tools. Pass `token_counter=` when numbers matter.
|
|
577
|
+
- **Detection isn't perfect.** Pattern rules miss unusual formats, and
|
|
578
|
+
GLiNER misses some names and companies (e.g. a company only mentioned as
|
|
579
|
+
"the X clinic"). Anything missed is sent as-is. Test on your own data.
|
|
580
|
+
- **Some answers get a little worse with redaction on.** The model can't
|
|
581
|
+
transliterate a hidden name, infer gender, or calculate with hidden money
|
|
582
|
+
amounts (which is why hiding money is opt-in); see
|
|
583
|
+
[the benchmark](#does-hiding-the-data-make-answers-worse).
|
|
584
|
+
- **Privacy costs a few tokens.** When something was hidden, the note to the
|
|
585
|
+
model (~45–120 tokens) and the longer stable placeholders add input tokens. On short prompts
|
|
586
|
+
that can outweigh trimming; `placeholder_hint=False` turns the note off
|
|
587
|
+
(not recommended for Claude).
|
|
588
|
+
- **The mapping lives in memory.** Keep the request's result until the
|
|
589
|
+
answer arrives; an answer handled after a restart can't be restored.
|
|
590
|
+
- **Compaction is lossy by design.** Summaries keep facts well with an API
|
|
591
|
+
summarizer, less well with a local 2B model, and pinned references keep
|
|
592
|
+
identifiers exact. Anything dropped is reported, not hidden.
|
|
593
|
+
- **Tool selection is lexical.** Paraphrased requests often fall back to
|
|
594
|
+
sending every tool: safe, but no saving.
|
|
595
|
+
- **Provider coverage:** Anthropic and Gemini have been tested live;
|
|
596
|
+
the OpenAI module is built from OpenAI's documentation but hasn't been
|
|
597
|
+
run against a real key yet.
|
|
598
|
+
- **Not included yet:** async clients and a PyPI
|
|
599
|
+
release (streamed answers can be restored with `StreamRestorer`). The API call is always your own synchronous function.
|
|
600
|
+
|
|
601
|
+
## Documentation
|
|
602
|
+
|
|
603
|
+
| Document | Contents |
|
|
604
|
+
|---|---|
|
|
605
|
+
| [docs/results.md](docs/results.md) | Every benchmark and live run, with methods |
|
|
606
|
+
| [experiments/privacy_quality/](experiments/privacy_quality/) | The answer-quality benchmark for redaction: cases, runner, how to reproduce |
|
|
607
|
+
| [docs/redaction.md](docs/redaction.md) | Detectors, backends, placeholders, names, emails, secrets, streaming |
|
|
608
|
+
| [docs/compaction.md](docs/compaction.md) | Stateless and rolling compaction, summaries, cache-aware mode, background summaries |
|
|
609
|
+
| [docs/tools-and-rag.md](docs/tools-and-rag.md) | Tool/MCP definition filtering, deferred loading, RAG chunk optimization |
|
|
610
|
+
| [docs/caching-and-providers.md](docs/caching-and-providers.md) | Prompt-caching structuring, per-provider details, other providers, live cache tests |
|
|
611
|
+
| [docs/measurement.md](docs/measurement.md) | Exact token counting and the savings log |
|
|
612
|
+
| [docs/testing.md](docs/testing.md) | Unit tests, demos, benchmarks, live tests, repository layout |
|
|
613
|
+
| [ROADMAP.md](ROADMAP.md) | Decisions, open questions and the full history of results |
|
|
614
|
+
|
|
615
|
+
Some documents cite research notes under `research/`; those notes aren't
|
|
616
|
+
published in this repository.
|
|
617
|
+
|
|
618
|
+
**Repository layout**
|
|
619
|
+
|
|
620
|
+
```
|
|
621
|
+
tonst/ the library
|
|
622
|
+
test_*.py unit tests (CI runs all of them)
|
|
623
|
+
experiments/ answer-quality benchmark for redaction (privacy_quality/)
|
|
624
|
+
examples/ runnable demos: demo.py (no API key needed), real_api_demo.py, cache_savings_demo_*.py per provider
|
|
625
|
+
benchmarks/ offline benchmark and live API tests (benchmark_free_features.py, live_test_*.py, benchmark_tonst.py)
|
|
626
|
+
scripts/research/ one-off diagnostic and tuning scripts behind the findings in docs/ (need GLiNER or Ollama)
|
|
627
|
+
docs/ detailed documentation
|
|
628
|
+
```
|
|
629
|
+
|
|
630
|
+
Run scripts from the repository root, e.g. `python3 examples/demo.py`.
|
|
631
|
+
API keys go in a `.env` file in the repository root (gitignored).
|
|
632
|
+
|
|
633
|
+
## Whitepaper, license and contributing
|
|
634
|
+
|
|
635
|
+
- **Whitepaper:** [Beyond the Prompt](https://claude.ai/artifact/2hcKTcfwBzAWev1PUGRv2x) · DOI [10.5281/zenodo.22745266](https://doi.org/10.5281/zenodo.22745266)
|
|
636
|
+
- **License:** MIT (see [LICENSE](LICENSE)).
|
|
637
|
+
- **Tests:** `pip install -r requirements-dev.txt && pytest` (315 tests, run on every push for Python 3.10–3.13). Live API tests and benchmarks are described in [docs/testing.md](docs/testing.md).
|
|
638
|
+
- Issues and pull requests are welcome.
|