contextpress 0.6.2__tar.gz → 0.6.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {contextpress-0.6.2 → contextpress-0.6.4}/CHANGELOG.md +19 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/PKG-INFO +21 -4
- {contextpress-0.6.2 → contextpress-0.6.4}/README.md +19 -2
- {contextpress-0.6.2 → contextpress-0.6.4}/ROADMAP.md +2 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/__init__.py +1 -1
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/core.py +17 -1
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/models.py +1 -1
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/normalizer.py +24 -10
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/stats.py +42 -3
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/budget.py +12 -3
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/filler.py +2 -11
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/structure.py +4 -2
- contextpress-0.6.4/contextpress/tools.py +181 -0
- contextpress-0.6.4/examples/langchain_roundtrip.py +39 -0
- contextpress-0.6.4/examples/openai_tools_compress.py +57 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/pyproject.toml +1 -1
- contextpress-0.6.4/tests/fixtures/chats/12_openai_tools.json +34 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/README.md +1 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_fixture_chats.py +1 -1
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v062.py +5 -1
- contextpress-0.6.4/tests/test_v063.py +101 -0
- contextpress-0.6.4/tests/test_v064.py +175 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/.gitignore +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/AGENTS.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/AUDIT.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/CITATION.cff +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/CONTRIBUTING.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/LICENSE +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/NOTICE +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/_bootstrap.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/compression.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/costs.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/_helpers.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/adapters.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/base.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/profiles.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/py.typed +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/registry.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/base.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/recency.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/repetition.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/resolution.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/text_sim.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/warnings_capture.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/agent_json_compress.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/agent_pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/benchmark_presets.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/dry_run_preview.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/estimate_and_stats.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_claude.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_gemini.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_ollama.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_openai.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/pick_preset.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/examples/structure_and_cost.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/01_filler_heavy.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/02_resolution_thread.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/03_repetition.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/04_long_history.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/05_agent_tools.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/06_rag_chunks.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/07_short_stable.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/08_mixed_ack_resolution.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/09_agent_tool_json.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/10_agent_repeated_logs.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/11_agent_mixed.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_budget.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_filler.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_llm_helpers.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_models.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_normalizer.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_recency.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_repetition.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_resolution.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_stats.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v03.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v04.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v05.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v051.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v052.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v053.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v054.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v056.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v058.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v060.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v061.py +0 -0
|
@@ -4,6 +4,25 @@ All notable changes to `contextpress` are recorded here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/)
|
|
5
5
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
6
|
|
|
7
|
+
## [0.6.4] - 2026-08-18
|
|
8
|
+
|
|
9
|
+
- **OpenAI tool messages** — ``role: tool`` / ``function`` are valid; ``tool_calls``,
|
|
10
|
+
``tool_call_id``, and ``name`` round-trip on dict messages.
|
|
11
|
+
- Structure minifies JSON in ``tool_calls[].function.arguments`` and tool-result content.
|
|
12
|
+
- Filler never drops empty assistant turns that only carry ``tool_calls``.
|
|
13
|
+
- Budget removes an assistant ``tool_calls`` turn together with matching ``role=tool``
|
|
14
|
+
results (no orphan tool messages).
|
|
15
|
+
- Example: `examples/openai_tools_compress.py`.
|
|
16
|
+
|
|
17
|
+
## [0.6.3] - 2026-08-12
|
|
18
|
+
|
|
19
|
+
- **LangChain round-trip** — ``compress()`` maps remaining turns back onto their original
|
|
20
|
+
message objects (not list index), so dropped turns no longer remap roles/content.
|
|
21
|
+
- **``output_tokens`` on cost stats** — ``attach_cost(output_tokens=...)``,
|
|
22
|
+
``compress(..., output_tokens=...)``, and ``ContextManager(cost_output_tokens=...)``
|
|
23
|
+
add assumed completion USD; ``summary()`` prints output + total when set.
|
|
24
|
+
- Example: `examples/langchain_roundtrip.py`.
|
|
25
|
+
|
|
7
26
|
## [0.6.2] - 2026-07-27
|
|
8
27
|
|
|
9
28
|
- **Agent-oriented fixtures** — three offline agent threads under `tests/fixtures/chats/`
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: contextpress
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.4
|
|
4
4
|
Summary: Deterministic context compression for LLM chat, RAG, and agent pipelines
|
|
5
5
|
Project-URL: Homepage, https://github.com/Taha-azizi/contextpress
|
|
6
6
|
Project-URL: Documentation, https://github.com/Taha-azizi/contextpress#readme
|
|
@@ -244,6 +244,8 @@ Description-Content-Type: text/markdown
|
|
|
244
244
|
Deterministic context compression for LLM chat, RAG, and agent pipelines.
|
|
245
245
|
Created and maintained by **[Taha Azizi](https://github.com/Taha-azizi)**.
|
|
246
246
|
|
|
247
|
+
**Write-up:** [Introducing contextpress](https://pub.towardsai.net/introducing-contextpress-the-python-library-that-refactors-your-llm-context-c57965617edb) — Towards AI (Medium)
|
|
248
|
+
|
|
247
249
|
---
|
|
248
250
|
|
|
249
251
|
## Project Status
|
|
@@ -389,7 +391,7 @@ That script builds a long history and a tight `token_budget` so you can see turn
|
|
|
389
391
|
|
|
390
392
|
- **chat** — Typical back-and-forth dialogue. Filler removal, repetition deduplication, resolution collapsing, recency weighting, and token budgets are tuned for conversational flow.
|
|
391
393
|
- **rag_doc** — Document chunks or RAG context. Resolution is off; repetition compares all chunks; recency uses relevance to the latest user query instead of chat recency.
|
|
392
|
-
- **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present.
|
|
394
|
+
- **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present. OpenAI Chat Completions ``tool_calls`` / ``role: tool`` messages are first-class (0.6.4+).
|
|
393
395
|
|
|
394
396
|
```python
|
|
395
397
|
ContextManager(type="chat")
|
|
@@ -398,6 +400,7 @@ ContextManager(type="agent")
|
|
|
398
400
|
```
|
|
399
401
|
|
|
400
402
|
Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.py).
|
|
403
|
+
OpenAI tools example: [`examples/openai_tools_compress.py`](examples/openai_tools_compress.py).
|
|
401
404
|
|
|
402
405
|
## Pipeline stages
|
|
403
406
|
|
|
@@ -406,7 +409,7 @@ Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.p
|
|
|
406
409
|
3. **Repetition** — TF-IDF cosine similarity; keeps the more recent of similar turns.
|
|
407
410
|
4. **Resolution** — Collapses agreed threads into a single `RESOLVED:` synthetic system turn (chat/agent only).
|
|
408
411
|
5. **Recency** — Extractively compresses older turns (or low-relevance chunks in `rag_doc`) while preserving the latest context.
|
|
409
|
-
6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns.
|
|
412
|
+
6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns. Assistant ``tool_calls`` and matching ``role: tool`` results are dropped together (0.6.4+).
|
|
410
413
|
|
|
411
414
|
**Cost estimate** (0.6+, approximate list prices for planning):
|
|
412
415
|
|
|
@@ -436,6 +439,20 @@ print(result.summary())
|
|
|
436
439
|
# est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
|
|
437
440
|
```
|
|
438
441
|
|
|
442
|
+
**Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
|
|
443
|
+
|
|
444
|
+
```python
|
|
445
|
+
cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
|
|
446
|
+
result = cm.compress(messages, token_budget=2000, return_stats=True)
|
|
447
|
+
print(result.summary())
|
|
448
|
+
# ...
|
|
449
|
+
# est. output cost: $0.000120 (200 tokens)
|
|
450
|
+
# est. total: $0.000246 -> $0.000181
|
|
451
|
+
```
|
|
452
|
+
|
|
453
|
+
LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
|
|
454
|
+
dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
|
|
455
|
+
|
|
439
456
|
See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
|
|
440
457
|
|
|
441
458
|
## Tier 1 vs Tier 2 (classical NLP vs LLM)
|
|
@@ -3,6 +3,8 @@
|
|
|
3
3
|
Deterministic context compression for LLM chat, RAG, and agent pipelines.
|
|
4
4
|
Created and maintained by **[Taha Azizi](https://github.com/Taha-azizi)**.
|
|
5
5
|
|
|
6
|
+
**Write-up:** [Introducing contextpress](https://pub.towardsai.net/introducing-contextpress-the-python-library-that-refactors-your-llm-context-c57965617edb) — Towards AI (Medium)
|
|
7
|
+
|
|
6
8
|
---
|
|
7
9
|
|
|
8
10
|
## Project Status
|
|
@@ -148,7 +150,7 @@ That script builds a long history and a tight `token_budget` so you can see turn
|
|
|
148
150
|
|
|
149
151
|
- **chat** — Typical back-and-forth dialogue. Filler removal, repetition deduplication, resolution collapsing, recency weighting, and token budgets are tuned for conversational flow.
|
|
150
152
|
- **rag_doc** — Document chunks or RAG context. Resolution is off; repetition compares all chunks; recency uses relevance to the latest user query instead of chat recency.
|
|
151
|
-
- **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present.
|
|
153
|
+
- **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present. OpenAI Chat Completions ``tool_calls`` / ``role: tool`` messages are first-class (0.6.4+).
|
|
152
154
|
|
|
153
155
|
```python
|
|
154
156
|
ContextManager(type="chat")
|
|
@@ -157,6 +159,7 @@ ContextManager(type="agent")
|
|
|
157
159
|
```
|
|
158
160
|
|
|
159
161
|
Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.py).
|
|
162
|
+
OpenAI tools example: [`examples/openai_tools_compress.py`](examples/openai_tools_compress.py).
|
|
160
163
|
|
|
161
164
|
## Pipeline stages
|
|
162
165
|
|
|
@@ -165,7 +168,7 @@ Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.p
|
|
|
165
168
|
3. **Repetition** — TF-IDF cosine similarity; keeps the more recent of similar turns.
|
|
166
169
|
4. **Resolution** — Collapses agreed threads into a single `RESOLVED:` synthetic system turn (chat/agent only).
|
|
167
170
|
5. **Recency** — Extractively compresses older turns (or low-relevance chunks in `rag_doc`) while preserving the latest context.
|
|
168
|
-
6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns.
|
|
171
|
+
6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns. Assistant ``tool_calls`` and matching ``role: tool`` results are dropped together (0.6.4+).
|
|
169
172
|
|
|
170
173
|
**Cost estimate** (0.6+, approximate list prices for planning):
|
|
171
174
|
|
|
@@ -195,6 +198,20 @@ print(result.summary())
|
|
|
195
198
|
# est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
|
|
196
199
|
```
|
|
197
200
|
|
|
201
|
+
**Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
|
|
205
|
+
result = cm.compress(messages, token_budget=2000, return_stats=True)
|
|
206
|
+
print(result.summary())
|
|
207
|
+
# ...
|
|
208
|
+
# est. output cost: $0.000120 (200 tokens)
|
|
209
|
+
# est. total: $0.000246 -> $0.000181
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
|
|
213
|
+
dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
|
|
214
|
+
|
|
198
215
|
See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
|
|
199
216
|
|
|
200
217
|
## Tier 1 vs Tier 2 (classical NLP vs LLM)
|
|
@@ -40,5 +40,7 @@ deterministic Tier‑1 NLP for chat / RAG / agent **message histories**, with op
|
|
|
40
40
|
| **0.6.0** | `structure` stage + `estimate_cost()` + this roadmap — shipped |
|
|
41
41
|
| **0.6.1** | Wire estimated USD into `CompressionStats` / reports — shipped |
|
|
42
42
|
| **0.6.2** | Agent-oriented fixtures for JSON/tool payloads; `summary()` report — shipped |
|
|
43
|
+
| **0.6.3** | LangChain compress round-trip; `output_tokens` on cost stats / `summary()` — shipped |
|
|
44
|
+
| **0.6.4** | OpenAI `tool_calls` / `role=tool` round-trip, JSON minify, budget pair integrity — shipped |
|
|
43
45
|
|
|
44
46
|
Stay classical-NLP-first; keep optional LLM extras optional.
|
|
@@ -48,6 +48,7 @@ class ContextManager:
|
|
|
48
48
|
llm_max_summary_tokens: int = 2048,
|
|
49
49
|
llm_mode: str = "replace_all",
|
|
50
50
|
cost_provider: str | None = None,
|
|
51
|
+
cost_output_tokens: int = 0,
|
|
51
52
|
):
|
|
52
53
|
if type not in PROFILES:
|
|
53
54
|
raise ValueError(f"unknown context type {type!r}")
|
|
@@ -65,6 +66,7 @@ class ContextManager:
|
|
|
65
66
|
self.llm_mode = llm_mode
|
|
66
67
|
# When set, compress(..., return_stats=True) attaches USD fields on stats.
|
|
67
68
|
self.cost_provider = cost_provider
|
|
69
|
+
self.cost_output_tokens = int(cost_output_tokens)
|
|
68
70
|
self._custom_stages: dict[str, StageConfig] = {}
|
|
69
71
|
|
|
70
72
|
def estimate_tokens(self, messages: Any, *, model: str | None = None) -> int:
|
|
@@ -185,6 +187,7 @@ class ContextManager:
|
|
|
185
187
|
return_stats: bool = False,
|
|
186
188
|
dry_run: bool = False,
|
|
187
189
|
cost_provider: str | None = None,
|
|
190
|
+
output_tokens: int | None = None,
|
|
188
191
|
) -> Any | CompressionResult:
|
|
189
192
|
"""Run the pipeline; return value matches input shape (dict list, tuples, strings, etc.).
|
|
190
193
|
|
|
@@ -193,6 +196,8 @@ class ContextManager:
|
|
|
193
196
|
With ``dry_run=True``, runs Tier 1 only (no LLM calls) and returns the original messages.
|
|
194
197
|
When ``cost_provider`` (or ``self.cost_provider``) is set and stats are returned,
|
|
195
198
|
``stats`` includes approximate input USD before/after compression.
|
|
199
|
+
``output_tokens`` (or ``self.cost_output_tokens``) adds an assumed completion cost
|
|
200
|
+
that is unchanged by compression.
|
|
196
201
|
"""
|
|
197
202
|
if dry_run:
|
|
198
203
|
return_stats = True
|
|
@@ -228,7 +233,14 @@ class ContextManager:
|
|
|
228
233
|
stats.warnings_emitted = captured
|
|
229
234
|
prov = cost_provider if cost_provider is not None else self.cost_provider
|
|
230
235
|
if prov is not None:
|
|
231
|
-
|
|
236
|
+
out_tok = (
|
|
237
|
+
output_tokens if output_tokens is not None else self.cost_output_tokens
|
|
238
|
+
)
|
|
239
|
+
stats.attach_cost(
|
|
240
|
+
provider=prov,
|
|
241
|
+
model=self.model or "gpt-4o-mini",
|
|
242
|
+
output_tokens=out_tok,
|
|
243
|
+
)
|
|
232
244
|
if dry_run:
|
|
233
245
|
messages_out = denormalize_output(clone_conversation(conv), ctx)
|
|
234
246
|
else:
|
|
@@ -249,6 +261,7 @@ class ContextManager:
|
|
|
249
261
|
return_stats: bool = False,
|
|
250
262
|
dry_run: bool = False,
|
|
251
263
|
cost_provider: str | None = None,
|
|
264
|
+
output_tokens: int | None = None,
|
|
252
265
|
) -> list[Any] | list[CompressionResult]:
|
|
253
266
|
"""Run ``compress()`` on each conversation in ``conversations``."""
|
|
254
267
|
if not isinstance(conversations, list):
|
|
@@ -263,6 +276,7 @@ class ContextManager:
|
|
|
263
276
|
return_stats=return_stats,
|
|
264
277
|
dry_run=dry_run,
|
|
265
278
|
cost_provider=cost_provider,
|
|
279
|
+
output_tokens=output_tokens,
|
|
266
280
|
)
|
|
267
281
|
for messages in conversations
|
|
268
282
|
]
|
|
@@ -278,6 +292,7 @@ class ContextManager:
|
|
|
278
292
|
return_stats: bool = False,
|
|
279
293
|
dry_run: bool = False,
|
|
280
294
|
cost_provider: str | None = None,
|
|
295
|
+
output_tokens: int | None = None,
|
|
281
296
|
) -> Any | CompressionResult:
|
|
282
297
|
"""Async wrapper around ``compress()`` (runs in a worker thread)."""
|
|
283
298
|
return await asyncio.to_thread(
|
|
@@ -290,6 +305,7 @@ class ContextManager:
|
|
|
290
305
|
return_stats=return_stats,
|
|
291
306
|
dry_run=dry_run,
|
|
292
307
|
cost_provider=cost_provider,
|
|
308
|
+
output_tokens=output_tokens,
|
|
293
309
|
)
|
|
294
310
|
|
|
295
311
|
def set_compression(self, compression: str) -> None:
|
|
@@ -25,7 +25,7 @@ class Turn:
|
|
|
25
25
|
This is the canonical unit the entire pipeline operates on.
|
|
26
26
|
"""
|
|
27
27
|
|
|
28
|
-
role: str # "user" | "assistant" | "system"
|
|
28
|
+
role: str # "user" | "assistant" | "system" | "tool" | "function"
|
|
29
29
|
content: str | list[ContentBlock] # string for simple, list for multimodal
|
|
30
30
|
timestamp: datetime | None = None
|
|
31
31
|
metadata: dict[str, Any] = field(default_factory=dict)
|
|
@@ -12,11 +12,14 @@ from typing import Any
|
|
|
12
12
|
|
|
13
13
|
from contextpress.models import ContentBlock, Conversation, Turn
|
|
14
14
|
|
|
15
|
-
_VALID_ROLES = frozenset({"user", "assistant", "system"})
|
|
15
|
+
_VALID_ROLES = frozenset({"user", "assistant", "system", "tool", "function"})
|
|
16
|
+
_TOOL_COPY_KEYS = ("tool_calls", "tool_call_id", "name")
|
|
16
17
|
_LC_TYPE_MAP = {
|
|
17
18
|
"human": "user",
|
|
18
19
|
"ai": "assistant",
|
|
19
20
|
"system": "system",
|
|
21
|
+
"tool": "tool",
|
|
22
|
+
"function": "function",
|
|
20
23
|
}
|
|
21
24
|
|
|
22
25
|
|
|
@@ -201,6 +204,9 @@ def normalize_messages(
|
|
|
201
204
|
ts = _parse_timestamp(d)
|
|
202
205
|
content = d.get("content", "")
|
|
203
206
|
meta: dict[str, Any] = {"_dict_index": i, "_original_dict": d}
|
|
207
|
+
for key in _TOOL_COPY_KEYS:
|
|
208
|
+
if key in d:
|
|
209
|
+
meta[key] = copy.deepcopy(d[key])
|
|
204
210
|
|
|
205
211
|
if isinstance(content, list):
|
|
206
212
|
blocks = _blocks_from_openai_style(content)
|
|
@@ -247,15 +253,14 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
|
|
|
247
253
|
return out
|
|
248
254
|
|
|
249
255
|
if fmt == "langchain":
|
|
250
|
-
# Reconstruct LangChain objects
|
|
251
|
-
|
|
252
|
-
if not lc_objs:
|
|
253
|
-
return []
|
|
256
|
+
# Reconstruct LangChain objects from the turn's original message, not list index
|
|
257
|
+
# (dropped turns would otherwise remap remaining content onto the wrong objects).
|
|
254
258
|
result = []
|
|
255
|
-
for
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
+
for t in turns:
|
|
260
|
+
orig = t.metadata.get("_lc_original") or t.metadata.get("_lc_obj")
|
|
261
|
+
text = _turn_to_plain_text(t)
|
|
262
|
+
if orig is not None:
|
|
263
|
+
obj = copy.copy(orig)
|
|
259
264
|
if hasattr(obj, "content"):
|
|
260
265
|
try:
|
|
261
266
|
obj.content = text
|
|
@@ -270,7 +275,7 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
|
|
|
270
275
|
self.type = role
|
|
271
276
|
self.content = content
|
|
272
277
|
|
|
273
|
-
result.append(_Msg(t.role,
|
|
278
|
+
result.append(_Msg(t.role, text))
|
|
274
279
|
return result
|
|
275
280
|
|
|
276
281
|
# dict_list
|
|
@@ -283,8 +288,17 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
|
|
|
283
288
|
base["role"] = t.role
|
|
284
289
|
if isinstance(t.content, list):
|
|
285
290
|
base["content"] = _blocks_to_openai_style(t.content)
|
|
291
|
+
elif (
|
|
292
|
+
t.content == ""
|
|
293
|
+
and isinstance(t.metadata.get("_original_dict"), dict)
|
|
294
|
+
and t.metadata["_original_dict"].get("content") is None
|
|
295
|
+
):
|
|
296
|
+
base["content"] = None
|
|
286
297
|
else:
|
|
287
298
|
base["content"] = t.content
|
|
299
|
+
for key in _TOOL_COPY_KEYS:
|
|
300
|
+
if key in t.metadata:
|
|
301
|
+
base[key] = copy.deepcopy(t.metadata[key])
|
|
288
302
|
out_dicts.append(base)
|
|
289
303
|
return out_dicts
|
|
290
304
|
|
|
@@ -10,6 +10,7 @@ import tiktoken
|
|
|
10
10
|
from contextpress.costs import estimate_token_cost
|
|
11
11
|
from contextpress.models import Conversation, Turn
|
|
12
12
|
from contextpress.normalizer import extract_text_for_processing
|
|
13
|
+
from contextpress.tools import tool_payload_text
|
|
13
14
|
|
|
14
15
|
|
|
15
16
|
def get_encoding(model: str | None) -> tiktoken.Encoding:
|
|
@@ -26,6 +27,9 @@ def count_turn_tokens(turn: Turn, encoding: tiktoken.Encoding) -> int:
|
|
|
26
27
|
body = turn.content
|
|
27
28
|
else:
|
|
28
29
|
body = extract_text_for_processing(turn)
|
|
30
|
+
extra = tool_payload_text(turn)
|
|
31
|
+
if extra:
|
|
32
|
+
body = f"{body}\n{extra}" if body else extra
|
|
29
33
|
return len(encoding.encode(f"{turn.role}\n{body}"))
|
|
30
34
|
|
|
31
35
|
|
|
@@ -57,6 +61,11 @@ class CompressionStats:
|
|
|
57
61
|
cost_model: str | None = None
|
|
58
62
|
estimated_input_cost_before_usd: float | None = None
|
|
59
63
|
estimated_input_cost_after_usd: float | None = None
|
|
64
|
+
# Optional completion-side estimate (0.6.3+); same before/after (compression is input-only)
|
|
65
|
+
estimated_output_tokens: int | None = None
|
|
66
|
+
estimated_output_cost_usd: float | None = None
|
|
67
|
+
estimated_total_cost_before_usd: float | None = None
|
|
68
|
+
estimated_total_cost_after_usd: float | None = None
|
|
60
69
|
|
|
61
70
|
@property
|
|
62
71
|
def turns_removed(self) -> int:
|
|
@@ -86,14 +95,32 @@ class CompressionStats:
|
|
|
86
95
|
*,
|
|
87
96
|
provider: str = "openai",
|
|
88
97
|
model: str | None = "gpt-4o-mini",
|
|
98
|
+
output_tokens: int = 0,
|
|
89
99
|
) -> CompressionStats:
|
|
90
|
-
"""Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self.
|
|
91
|
-
|
|
92
|
-
|
|
100
|
+
"""Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self.
|
|
101
|
+
|
|
102
|
+
``output_tokens`` is an assumed completion size (unchanged by compression).
|
|
103
|
+
"""
|
|
104
|
+
before = estimate_token_cost(
|
|
105
|
+
self.tokens_before, provider=provider, model=model, output_tokens=output_tokens
|
|
106
|
+
)
|
|
107
|
+
after = estimate_token_cost(
|
|
108
|
+
self.tokens_after, provider=provider, model=model, output_tokens=output_tokens
|
|
109
|
+
)
|
|
93
110
|
self.cost_provider = before.provider
|
|
94
111
|
self.cost_model = before.model
|
|
95
112
|
self.estimated_input_cost_before_usd = before.input_cost_usd
|
|
96
113
|
self.estimated_input_cost_after_usd = after.input_cost_usd
|
|
114
|
+
if output_tokens > 0:
|
|
115
|
+
self.estimated_output_tokens = after.output_tokens
|
|
116
|
+
self.estimated_output_cost_usd = after.output_cost_usd
|
|
117
|
+
self.estimated_total_cost_before_usd = before.total_cost_usd
|
|
118
|
+
self.estimated_total_cost_after_usd = after.total_cost_usd
|
|
119
|
+
else:
|
|
120
|
+
self.estimated_output_tokens = None
|
|
121
|
+
self.estimated_output_cost_usd = None
|
|
122
|
+
self.estimated_total_cost_before_usd = None
|
|
123
|
+
self.estimated_total_cost_after_usd = None
|
|
97
124
|
return self
|
|
98
125
|
|
|
99
126
|
def summary(self) -> str:
|
|
@@ -118,6 +145,14 @@ class CompressionStats:
|
|
|
118
145
|
"est. input cost: "
|
|
119
146
|
f"${before_usd:.6f} -> ${after_usd:.6f} (saved ${saved:.6f})"
|
|
120
147
|
)
|
|
148
|
+
out_usd = self.estimated_output_cost_usd
|
|
149
|
+
out_tok = self.estimated_output_tokens
|
|
150
|
+
if out_usd is not None and out_tok is not None:
|
|
151
|
+
lines.append(f"est. output cost: ${out_usd:.6f} ({out_tok} tokens)")
|
|
152
|
+
total_before = self.estimated_total_cost_before_usd
|
|
153
|
+
total_after = self.estimated_total_cost_after_usd
|
|
154
|
+
if total_before is not None and total_after is not None:
|
|
155
|
+
lines.append(f"est. total: ${total_before:.6f} -> ${total_after:.6f}")
|
|
121
156
|
return "\n".join(lines)
|
|
122
157
|
|
|
123
158
|
def to_dict(self) -> dict[str, Any]:
|
|
@@ -145,6 +180,10 @@ class CompressionStats:
|
|
|
145
180
|
"estimated_input_cost_before_usd": self.estimated_input_cost_before_usd,
|
|
146
181
|
"estimated_input_cost_after_usd": self.estimated_input_cost_after_usd,
|
|
147
182
|
"estimated_cost_saved_usd": self.estimated_cost_saved_usd,
|
|
183
|
+
"estimated_output_tokens": self.estimated_output_tokens,
|
|
184
|
+
"estimated_output_cost_usd": self.estimated_output_cost_usd,
|
|
185
|
+
"estimated_total_cost_before_usd": self.estimated_total_cost_before_usd,
|
|
186
|
+
"estimated_total_cost_after_usd": self.estimated_total_cost_after_usd,
|
|
148
187
|
}
|
|
149
188
|
|
|
150
189
|
|
|
@@ -8,6 +8,7 @@ import tiktoken
|
|
|
8
8
|
from contextpress.models import Conversation, Turn
|
|
9
9
|
from contextpress.stats import count_turn_tokens, get_encoding
|
|
10
10
|
from contextpress.strategies.base import BaseStrategy
|
|
11
|
+
from contextpress.tools import tool_group_indices
|
|
11
12
|
|
|
12
13
|
|
|
13
14
|
def _truncate_system_turn(turn: Turn, encoding: tiktoken.Encoding, max_tokens: int) -> Turn:
|
|
@@ -73,9 +74,17 @@ class BudgetStrategy(BaseStrategy):
|
|
|
73
74
|
keep = min(2, len(ns_positions))
|
|
74
75
|
protected = set(ns_positions[-keep:]) if keep else set()
|
|
75
76
|
removable = [i for i in ns_positions if i not in protected]
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
77
|
+
group: list[int] | None = None
|
|
78
|
+
for idx in removable:
|
|
79
|
+
candidate = tool_group_indices(turns, idx)
|
|
80
|
+
if any(g in protected for g in candidate):
|
|
81
|
+
continue
|
|
82
|
+
group = candidate
|
|
83
|
+
break
|
|
84
|
+
if group is not None:
|
|
85
|
+
for g in sorted(group, reverse=True):
|
|
86
|
+
turns.pop(g)
|
|
87
|
+
n_removed += 1
|
|
79
88
|
continue
|
|
80
89
|
|
|
81
90
|
# Last resort: truncate system (see behavior contract note on invariant 1)
|
|
@@ -6,6 +6,7 @@ import re
|
|
|
6
6
|
from contextpress.models import Conversation, Turn
|
|
7
7
|
from contextpress.normalizer import apply_text_to_turn, extract_text_for_processing
|
|
8
8
|
from contextpress.strategies.base import BaseStrategy
|
|
9
|
+
from contextpress.tools import has_tool_marker
|
|
9
10
|
|
|
10
11
|
# Curated lists — longer phrases first for safe replacement order
|
|
11
12
|
FILLER_PHRASES = [
|
|
@@ -70,16 +71,6 @@ ACKNOWLEDGEMENT_PHRASES = [
|
|
|
70
71
|
"thank you for that",
|
|
71
72
|
]
|
|
72
73
|
|
|
73
|
-
_TOOL_MARKERS = ("tool_calls", "tool_call", "tool_use", "tool_result", "<tool", "[tool")
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
def _has_tool_marker(turn: Turn) -> bool:
|
|
77
|
-
meta = turn.metadata or {}
|
|
78
|
-
if any(k in meta for k in ("tool_calls", "tool_call", "tool_use", "tool_result")):
|
|
79
|
-
return True
|
|
80
|
-
text = extract_text_for_processing(turn).lower()
|
|
81
|
-
return any(m in text for m in _TOOL_MARKERS)
|
|
82
|
-
|
|
83
74
|
|
|
84
75
|
def _build_filler_pattern() -> re.Pattern[str]:
|
|
85
76
|
pattern_parts: list[str] = []
|
|
@@ -155,7 +146,7 @@ class FillerStrategy(BaseStrategy):
|
|
|
155
146
|
new_turns.append(nt)
|
|
156
147
|
continue
|
|
157
148
|
|
|
158
|
-
if
|
|
149
|
+
if has_tool_marker(turn):
|
|
159
150
|
new_turns.append(copy.deepcopy(turn))
|
|
160
151
|
continue
|
|
161
152
|
|
|
@@ -9,6 +9,7 @@ import re
|
|
|
9
9
|
from contextpress.models import Conversation, Turn
|
|
10
10
|
from contextpress.normalizer import apply_text_to_turn, extract_text_for_processing
|
|
11
11
|
from contextpress.strategies.base import BaseStrategy
|
|
12
|
+
from contextpress.tools import minify_tool_fields
|
|
12
13
|
|
|
13
14
|
_CODE_FENCE = re.compile(r"(```[\s\S]*?```)", re.MULTILINE)
|
|
14
15
|
_MULTI_BLANK = re.compile(r"\n{3,}")
|
|
@@ -85,9 +86,10 @@ class StructureStrategy(BaseStrategy):
|
|
|
85
86
|
text = extract_text_for_processing(turn)
|
|
86
87
|
compacted = compact_structure_text(text, aggressiveness=self.aggressiveness)
|
|
87
88
|
if compacted != text:
|
|
88
|
-
|
|
89
|
+
nt = apply_text_to_turn(turn, compacted)
|
|
89
90
|
else:
|
|
90
|
-
|
|
91
|
+
nt = copy.deepcopy(turn)
|
|
92
|
+
new_turns.append(minify_tool_fields(nt))
|
|
91
93
|
return Conversation(
|
|
92
94
|
turns=new_turns,
|
|
93
95
|
type=conversation.type,
|