contextpress 0.6.2__tar.gz → 0.6.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. {contextpress-0.6.2 → contextpress-0.6.4}/CHANGELOG.md +19 -0
  2. {contextpress-0.6.2 → contextpress-0.6.4}/PKG-INFO +21 -4
  3. {contextpress-0.6.2 → contextpress-0.6.4}/README.md +19 -2
  4. {contextpress-0.6.2 → contextpress-0.6.4}/ROADMAP.md +2 -0
  5. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/__init__.py +1 -1
  6. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/core.py +17 -1
  7. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/models.py +1 -1
  8. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/normalizer.py +24 -10
  9. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/stats.py +42 -3
  10. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/budget.py +12 -3
  11. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/filler.py +2 -11
  12. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/structure.py +4 -2
  13. contextpress-0.6.4/contextpress/tools.py +181 -0
  14. contextpress-0.6.4/examples/langchain_roundtrip.py +39 -0
  15. contextpress-0.6.4/examples/openai_tools_compress.py +57 -0
  16. {contextpress-0.6.2 → contextpress-0.6.4}/pyproject.toml +1 -1
  17. contextpress-0.6.4/tests/fixtures/chats/12_openai_tools.json +34 -0
  18. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/README.md +1 -0
  19. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_fixture_chats.py +1 -1
  20. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v062.py +5 -1
  21. contextpress-0.6.4/tests/test_v063.py +101 -0
  22. contextpress-0.6.4/tests/test_v064.py +175 -0
  23. {contextpress-0.6.2 → contextpress-0.6.4}/.gitignore +0 -0
  24. {contextpress-0.6.2 → contextpress-0.6.4}/AGENTS.md +0 -0
  25. {contextpress-0.6.2 → contextpress-0.6.4}/AUDIT.md +0 -0
  26. {contextpress-0.6.2 → contextpress-0.6.4}/CITATION.cff +0 -0
  27. {contextpress-0.6.2 → contextpress-0.6.4}/CONTRIBUTING.md +0 -0
  28. {contextpress-0.6.2 → contextpress-0.6.4}/LICENSE +0 -0
  29. {contextpress-0.6.2 → contextpress-0.6.4}/NOTICE +0 -0
  30. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/_bootstrap.py +0 -0
  31. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/compression.py +0 -0
  32. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/costs.py +0 -0
  33. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/__init__.py +0 -0
  34. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/_helpers.py +0 -0
  35. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/adapters.py +0 -0
  36. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/llm/base.py +0 -0
  37. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/pipeline.py +0 -0
  38. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/profiles.py +0 -0
  39. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/py.typed +0 -0
  40. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/registry.py +0 -0
  41. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/__init__.py +0 -0
  42. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/base.py +0 -0
  43. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/recency.py +0 -0
  44. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/repetition.py +0 -0
  45. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/strategies/resolution.py +0 -0
  46. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/text_sim.py +0 -0
  47. {contextpress-0.6.2 → contextpress-0.6.4}/contextpress/warnings_capture.py +0 -0
  48. {contextpress-0.6.2 → contextpress-0.6.4}/examples/agent_json_compress.py +0 -0
  49. {contextpress-0.6.2 → contextpress-0.6.4}/examples/agent_pipeline.py +0 -0
  50. {contextpress-0.6.2 → contextpress-0.6.4}/examples/benchmark_presets.py +0 -0
  51. {contextpress-0.6.2 → contextpress-0.6.4}/examples/dry_run_preview.py +0 -0
  52. {contextpress-0.6.2 → contextpress-0.6.4}/examples/estimate_and_stats.py +0 -0
  53. {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_claude.py +0 -0
  54. {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_gemini.py +0 -0
  55. {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_ollama.py +0 -0
  56. {contextpress-0.6.2 → contextpress-0.6.4}/examples/llm_tier_openai.py +0 -0
  57. {contextpress-0.6.2 → contextpress-0.6.4}/examples/pick_preset.py +0 -0
  58. {contextpress-0.6.2 → contextpress-0.6.4}/examples/structure_and_cost.py +0 -0
  59. {contextpress-0.6.2 → contextpress-0.6.4}/tests/__init__.py +0 -0
  60. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/01_filler_heavy.json +0 -0
  61. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/02_resolution_thread.json +0 -0
  62. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/03_repetition.json +0 -0
  63. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/04_long_history.json +0 -0
  64. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/05_agent_tools.json +0 -0
  65. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/06_rag_chunks.json +0 -0
  66. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/07_short_stable.json +0 -0
  67. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/08_mixed_ack_resolution.json +0 -0
  68. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/09_agent_tool_json.json +0 -0
  69. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/10_agent_repeated_logs.json +0 -0
  70. {contextpress-0.6.2 → contextpress-0.6.4}/tests/fixtures/chats/11_agent_mixed.json +0 -0
  71. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_budget.py +0 -0
  72. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_filler.py +0 -0
  73. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_llm_helpers.py +0 -0
  74. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_models.py +0 -0
  75. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_normalizer.py +0 -0
  76. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_pipeline.py +0 -0
  77. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_recency.py +0 -0
  78. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_repetition.py +0 -0
  79. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_resolution.py +0 -0
  80. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_stats.py +0 -0
  81. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v03.py +0 -0
  82. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v04.py +0 -0
  83. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v05.py +0 -0
  84. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v051.py +0 -0
  85. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v052.py +0 -0
  86. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v053.py +0 -0
  87. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v054.py +0 -0
  88. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v056.py +0 -0
  89. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v058.py +0 -0
  90. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v060.py +0 -0
  91. {contextpress-0.6.2 → contextpress-0.6.4}/tests/test_v061.py +0 -0
@@ -4,6 +4,25 @@ All notable changes to `contextpress` are recorded here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/)
5
5
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
6
6
 
7
+ ## [0.6.4] - 2026-08-18
8
+
9
+ - **OpenAI tool messages** — ``role: tool`` / ``function`` are valid; ``tool_calls``,
10
+ ``tool_call_id``, and ``name`` round-trip on dict messages.
11
+ - Structure minifies JSON in ``tool_calls[].function.arguments`` and tool-result content.
12
+ - Filler never drops empty assistant turns that only carry ``tool_calls``.
13
+ - Budget removes an assistant ``tool_calls`` turn together with matching ``role=tool``
14
+ results (no orphan tool messages).
15
+ - Example: `examples/openai_tools_compress.py`.
16
+
17
+ ## [0.6.3] - 2026-08-12
18
+
19
+ - **LangChain round-trip** — ``compress()`` maps remaining turns back onto their original
20
+ message objects (not list index), so dropped turns no longer remap roles/content.
21
+ - **``output_tokens`` on cost stats** — ``attach_cost(output_tokens=...)``,
22
+ ``compress(..., output_tokens=...)``, and ``ContextManager(cost_output_tokens=...)``
23
+ add assumed completion USD; ``summary()`` prints output + total when set.
24
+ - Example: `examples/langchain_roundtrip.py`.
25
+
7
26
  ## [0.6.2] - 2026-07-27
8
27
 
9
28
  - **Agent-oriented fixtures** — three offline agent threads under `tests/fixtures/chats/`
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: contextpress
3
- Version: 0.6.2
3
+ Version: 0.6.4
4
4
  Summary: Deterministic context compression for LLM chat, RAG, and agent pipelines
5
5
  Project-URL: Homepage, https://github.com/Taha-azizi/contextpress
6
6
  Project-URL: Documentation, https://github.com/Taha-azizi/contextpress#readme
@@ -244,6 +244,8 @@ Description-Content-Type: text/markdown
244
244
  Deterministic context compression for LLM chat, RAG, and agent pipelines.
245
245
  Created and maintained by **[Taha Azizi](https://github.com/Taha-azizi)**.
246
246
 
247
+ **Write-up:** [Introducing contextpress](https://pub.towardsai.net/introducing-contextpress-the-python-library-that-refactors-your-llm-context-c57965617edb) — Towards AI (Medium)
248
+
247
249
  ---
248
250
 
249
251
  ## Project Status
@@ -389,7 +391,7 @@ That script builds a long history and a tight `token_budget` so you can see turn
389
391
 
390
392
  - **chat** — Typical back-and-forth dialogue. Filler removal, repetition deduplication, resolution collapsing, recency weighting, and token budgets are tuned for conversational flow.
391
393
  - **rag_doc** — Document chunks or RAG context. Resolution is off; repetition compares all chunks; recency uses relevance to the latest user query instead of chat recency.
392
- - **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present.
394
+ - **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present. OpenAI Chat Completions ``tool_calls`` / ``role: tool`` messages are first-class (0.6.4+).
393
395
 
394
396
  ```python
395
397
  ContextManager(type="chat")
@@ -398,6 +400,7 @@ ContextManager(type="agent")
398
400
  ```
399
401
 
400
402
  Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.py).
403
+ OpenAI tools example: [`examples/openai_tools_compress.py`](examples/openai_tools_compress.py).
401
404
 
402
405
  ## Pipeline stages
403
406
 
@@ -406,7 +409,7 @@ Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.p
406
409
  3. **Repetition** — TF-IDF cosine similarity; keeps the more recent of similar turns.
407
410
  4. **Resolution** — Collapses agreed threads into a single `RESOLVED:` synthetic system turn (chat/agent only).
408
411
  5. **Recency** — Extractively compresses older turns (or low-relevance chunks in `rag_doc`) while preserving the latest context.
409
- 6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns.
412
+ 6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns. Assistant ``tool_calls`` and matching ``role: tool`` results are dropped together (0.6.4+).
410
413
 
411
414
  **Cost estimate** (0.6+, approximate list prices for planning):
412
415
 
@@ -436,6 +439,20 @@ print(result.summary())
436
439
  # est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
437
440
  ```
438
441
 
442
+ **Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
443
+
444
+ ```python
445
+ cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
446
+ result = cm.compress(messages, token_budget=2000, return_stats=True)
447
+ print(result.summary())
448
+ # ...
449
+ # est. output cost: $0.000120 (200 tokens)
450
+ # est. total: $0.000246 -> $0.000181
451
+ ```
452
+
453
+ LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
454
+ dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
455
+
439
456
  See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
440
457
 
441
458
  ## Tier 1 vs Tier 2 (classical NLP vs LLM)
@@ -3,6 +3,8 @@
3
3
  Deterministic context compression for LLM chat, RAG, and agent pipelines.
4
4
  Created and maintained by **[Taha Azizi](https://github.com/Taha-azizi)**.
5
5
 
6
+ **Write-up:** [Introducing contextpress](https://pub.towardsai.net/introducing-contextpress-the-python-library-that-refactors-your-llm-context-c57965617edb) — Towards AI (Medium)
7
+
6
8
  ---
7
9
 
8
10
  ## Project Status
@@ -148,7 +150,7 @@ That script builds a long history and a tight `token_budget` so you can see turn
148
150
 
149
151
  - **chat** — Typical back-and-forth dialogue. Filler removal, repetition deduplication, resolution collapsing, recency weighting, and token budgets are tuned for conversational flow.
150
152
  - **rag_doc** — Document chunks or RAG context. Resolution is off; repetition compares all chunks; recency uses relevance to the latest user query instead of chat recency.
151
- - **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present.
153
+ - **agent** — Tool-using or task-oriented threads. Resolution can trigger on a single high-confidence completion signal; filler rules preserve tool-related turns when markers are present. OpenAI Chat Completions ``tool_calls`` / ``role: tool`` messages are first-class (0.6.4+).
152
154
 
153
155
  ```python
154
156
  ContextManager(type="chat")
@@ -157,6 +159,7 @@ ContextManager(type="agent")
157
159
  ```
158
160
 
159
161
  Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.py).
162
+ OpenAI tools example: [`examples/openai_tools_compress.py`](examples/openai_tools_compress.py).
160
163
 
161
164
  ## Pipeline stages
162
165
 
@@ -165,7 +168,7 @@ Runnable agent example: [`examples/agent_pipeline.py`](examples/agent_pipeline.p
165
168
  3. **Repetition** — TF-IDF cosine similarity; keeps the more recent of similar turns.
166
169
  4. **Resolution** — Collapses agreed threads into a single `RESOLVED:` synthetic system turn (chat/agent only).
167
170
  5. **Recency** — Extractively compresses older turns (or low-relevance chunks in `rag_doc`) while preserving the latest context.
168
- 6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns.
171
+ 6. **Budget** — Enforces a hard token limit with `tiktoken`, removing oldest turns first while protecting system prompts and recent turns. Assistant ``tool_calls`` and matching ``role: tool`` results are dropped together (0.6.4+).
169
172
 
170
173
  **Cost estimate** (0.6+, approximate list prices for planning):
171
174
 
@@ -195,6 +198,20 @@ print(result.summary())
195
198
  # est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
196
199
  ```
197
200
 
201
+ **Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
202
+
203
+ ```python
204
+ cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
205
+ result = cm.compress(messages, token_budget=2000, return_stats=True)
206
+ print(result.summary())
207
+ # ...
208
+ # est. output cost: $0.000120 (200 tokens)
209
+ # est. total: $0.000246 -> $0.000181
210
+ ```
211
+
212
+ LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
213
+ dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
214
+
198
215
  See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
199
216
 
200
217
  ## Tier 1 vs Tier 2 (classical NLP vs LLM)
@@ -40,5 +40,7 @@ deterministic Tier‑1 NLP for chat / RAG / agent **message histories**, with op
40
40
  | **0.6.0** | `structure` stage + `estimate_cost()` + this roadmap — shipped |
41
41
  | **0.6.1** | Wire estimated USD into `CompressionStats` / reports — shipped |
42
42
  | **0.6.2** | Agent-oriented fixtures for JSON/tool payloads; `summary()` report — shipped |
43
+ | **0.6.3** | LangChain compress round-trip; `output_tokens` on cost stats / `summary()` — shipped |
44
+ | **0.6.4** | OpenAI `tool_calls` / `role=tool` round-trip, JSON minify, budget pair integrity — shipped |
43
45
 
44
46
  Stay classical-NLP-first; keep optional LLM extras optional.
@@ -16,7 +16,7 @@ __all__ = [
16
16
  "CompressionResult",
17
17
  "CompressionStats",
18
18
  ]
19
- __version__ = "0.6.2"
19
+ __version__ = "0.6.4"
20
20
 
21
21
 
22
22
  def __getattr__(name: str) -> Any:
@@ -48,6 +48,7 @@ class ContextManager:
48
48
  llm_max_summary_tokens: int = 2048,
49
49
  llm_mode: str = "replace_all",
50
50
  cost_provider: str | None = None,
51
+ cost_output_tokens: int = 0,
51
52
  ):
52
53
  if type not in PROFILES:
53
54
  raise ValueError(f"unknown context type {type!r}")
@@ -65,6 +66,7 @@ class ContextManager:
65
66
  self.llm_mode = llm_mode
66
67
  # When set, compress(..., return_stats=True) attaches USD fields on stats.
67
68
  self.cost_provider = cost_provider
69
+ self.cost_output_tokens = int(cost_output_tokens)
68
70
  self._custom_stages: dict[str, StageConfig] = {}
69
71
 
70
72
  def estimate_tokens(self, messages: Any, *, model: str | None = None) -> int:
@@ -185,6 +187,7 @@ class ContextManager:
185
187
  return_stats: bool = False,
186
188
  dry_run: bool = False,
187
189
  cost_provider: str | None = None,
190
+ output_tokens: int | None = None,
188
191
  ) -> Any | CompressionResult:
189
192
  """Run the pipeline; return value matches input shape (dict list, tuples, strings, etc.).
190
193
 
@@ -193,6 +196,8 @@ class ContextManager:
193
196
  With ``dry_run=True``, runs Tier 1 only (no LLM calls) and returns the original messages.
194
197
  When ``cost_provider`` (or ``self.cost_provider``) is set and stats are returned,
195
198
  ``stats`` includes approximate input USD before/after compression.
199
+ ``output_tokens`` (or ``self.cost_output_tokens``) adds an assumed completion cost
200
+ that is unchanged by compression.
196
201
  """
197
202
  if dry_run:
198
203
  return_stats = True
@@ -228,7 +233,14 @@ class ContextManager:
228
233
  stats.warnings_emitted = captured
229
234
  prov = cost_provider if cost_provider is not None else self.cost_provider
230
235
  if prov is not None:
231
- stats.attach_cost(provider=prov, model=self.model or "gpt-4o-mini")
236
+ out_tok = (
237
+ output_tokens if output_tokens is not None else self.cost_output_tokens
238
+ )
239
+ stats.attach_cost(
240
+ provider=prov,
241
+ model=self.model or "gpt-4o-mini",
242
+ output_tokens=out_tok,
243
+ )
232
244
  if dry_run:
233
245
  messages_out = denormalize_output(clone_conversation(conv), ctx)
234
246
  else:
@@ -249,6 +261,7 @@ class ContextManager:
249
261
  return_stats: bool = False,
250
262
  dry_run: bool = False,
251
263
  cost_provider: str | None = None,
264
+ output_tokens: int | None = None,
252
265
  ) -> list[Any] | list[CompressionResult]:
253
266
  """Run ``compress()`` on each conversation in ``conversations``."""
254
267
  if not isinstance(conversations, list):
@@ -263,6 +276,7 @@ class ContextManager:
263
276
  return_stats=return_stats,
264
277
  dry_run=dry_run,
265
278
  cost_provider=cost_provider,
279
+ output_tokens=output_tokens,
266
280
  )
267
281
  for messages in conversations
268
282
  ]
@@ -278,6 +292,7 @@ class ContextManager:
278
292
  return_stats: bool = False,
279
293
  dry_run: bool = False,
280
294
  cost_provider: str | None = None,
295
+ output_tokens: int | None = None,
281
296
  ) -> Any | CompressionResult:
282
297
  """Async wrapper around ``compress()`` (runs in a worker thread)."""
283
298
  return await asyncio.to_thread(
@@ -290,6 +305,7 @@ class ContextManager:
290
305
  return_stats=return_stats,
291
306
  dry_run=dry_run,
292
307
  cost_provider=cost_provider,
308
+ output_tokens=output_tokens,
293
309
  )
294
310
 
295
311
  def set_compression(self, compression: str) -> None:
@@ -25,7 +25,7 @@ class Turn:
25
25
  This is the canonical unit the entire pipeline operates on.
26
26
  """
27
27
 
28
- role: str # "user" | "assistant" | "system"
28
+ role: str # "user" | "assistant" | "system" | "tool" | "function"
29
29
  content: str | list[ContentBlock] # string for simple, list for multimodal
30
30
  timestamp: datetime | None = None
31
31
  metadata: dict[str, Any] = field(default_factory=dict)
@@ -12,11 +12,14 @@ from typing import Any
12
12
 
13
13
  from contextpress.models import ContentBlock, Conversation, Turn
14
14
 
15
- _VALID_ROLES = frozenset({"user", "assistant", "system"})
15
+ _VALID_ROLES = frozenset({"user", "assistant", "system", "tool", "function"})
16
+ _TOOL_COPY_KEYS = ("tool_calls", "tool_call_id", "name")
16
17
  _LC_TYPE_MAP = {
17
18
  "human": "user",
18
19
  "ai": "assistant",
19
20
  "system": "system",
21
+ "tool": "tool",
22
+ "function": "function",
20
23
  }
21
24
 
22
25
 
@@ -201,6 +204,9 @@ def normalize_messages(
201
204
  ts = _parse_timestamp(d)
202
205
  content = d.get("content", "")
203
206
  meta: dict[str, Any] = {"_dict_index": i, "_original_dict": d}
207
+ for key in _TOOL_COPY_KEYS:
208
+ if key in d:
209
+ meta[key] = copy.deepcopy(d[key])
204
210
 
205
211
  if isinstance(content, list):
206
212
  blocks = _blocks_from_openai_style(content)
@@ -247,15 +253,14 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
247
253
  return out
248
254
 
249
255
  if fmt == "langchain":
250
- # Reconstruct LangChain objects by copying original and setting content
251
- lc_objs = ctx.get("lc_objects", [])
252
- if not lc_objs:
253
- return []
256
+ # Reconstruct LangChain objects from the turn's original message, not list index
257
+ # (dropped turns would otherwise remap remaining content onto the wrong objects).
254
258
  result = []
255
- for i, t in enumerate(turns):
256
- if i < len(lc_objs):
257
- obj = copy.copy(lc_objs[i])
258
- text = _turn_to_plain_text(t)
259
+ for t in turns:
260
+ orig = t.metadata.get("_lc_original") or t.metadata.get("_lc_obj")
261
+ text = _turn_to_plain_text(t)
262
+ if orig is not None:
263
+ obj = copy.copy(orig)
259
264
  if hasattr(obj, "content"):
260
265
  try:
261
266
  obj.content = text
@@ -270,7 +275,7 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
270
275
  self.type = role
271
276
  self.content = content
272
277
 
273
- result.append(_Msg(t.role, _turn_to_plain_text(t)))
278
+ result.append(_Msg(t.role, text))
274
279
  return result
275
280
 
276
281
  # dict_list
@@ -283,8 +288,17 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
283
288
  base["role"] = t.role
284
289
  if isinstance(t.content, list):
285
290
  base["content"] = _blocks_to_openai_style(t.content)
291
+ elif (
292
+ t.content == ""
293
+ and isinstance(t.metadata.get("_original_dict"), dict)
294
+ and t.metadata["_original_dict"].get("content") is None
295
+ ):
296
+ base["content"] = None
286
297
  else:
287
298
  base["content"] = t.content
299
+ for key in _TOOL_COPY_KEYS:
300
+ if key in t.metadata:
301
+ base[key] = copy.deepcopy(t.metadata[key])
288
302
  out_dicts.append(base)
289
303
  return out_dicts
290
304
 
@@ -10,6 +10,7 @@ import tiktoken
10
10
  from contextpress.costs import estimate_token_cost
11
11
  from contextpress.models import Conversation, Turn
12
12
  from contextpress.normalizer import extract_text_for_processing
13
+ from contextpress.tools import tool_payload_text
13
14
 
14
15
 
15
16
  def get_encoding(model: str | None) -> tiktoken.Encoding:
@@ -26,6 +27,9 @@ def count_turn_tokens(turn: Turn, encoding: tiktoken.Encoding) -> int:
26
27
  body = turn.content
27
28
  else:
28
29
  body = extract_text_for_processing(turn)
30
+ extra = tool_payload_text(turn)
31
+ if extra:
32
+ body = f"{body}\n{extra}" if body else extra
29
33
  return len(encoding.encode(f"{turn.role}\n{body}"))
30
34
 
31
35
 
@@ -57,6 +61,11 @@ class CompressionStats:
57
61
  cost_model: str | None = None
58
62
  estimated_input_cost_before_usd: float | None = None
59
63
  estimated_input_cost_after_usd: float | None = None
64
+ # Optional completion-side estimate (0.6.3+); same before/after (compression is input-only)
65
+ estimated_output_tokens: int | None = None
66
+ estimated_output_cost_usd: float | None = None
67
+ estimated_total_cost_before_usd: float | None = None
68
+ estimated_total_cost_after_usd: float | None = None
60
69
 
61
70
  @property
62
71
  def turns_removed(self) -> int:
@@ -86,14 +95,32 @@ class CompressionStats:
86
95
  *,
87
96
  provider: str = "openai",
88
97
  model: str | None = "gpt-4o-mini",
98
+ output_tokens: int = 0,
89
99
  ) -> CompressionStats:
90
- """Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self."""
91
- before = estimate_token_cost(self.tokens_before, provider=provider, model=model)
92
- after = estimate_token_cost(self.tokens_after, provider=provider, model=model)
100
+ """Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self.
101
+
102
+ ``output_tokens`` is an assumed completion size (unchanged by compression).
103
+ """
104
+ before = estimate_token_cost(
105
+ self.tokens_before, provider=provider, model=model, output_tokens=output_tokens
106
+ )
107
+ after = estimate_token_cost(
108
+ self.tokens_after, provider=provider, model=model, output_tokens=output_tokens
109
+ )
93
110
  self.cost_provider = before.provider
94
111
  self.cost_model = before.model
95
112
  self.estimated_input_cost_before_usd = before.input_cost_usd
96
113
  self.estimated_input_cost_after_usd = after.input_cost_usd
114
+ if output_tokens > 0:
115
+ self.estimated_output_tokens = after.output_tokens
116
+ self.estimated_output_cost_usd = after.output_cost_usd
117
+ self.estimated_total_cost_before_usd = before.total_cost_usd
118
+ self.estimated_total_cost_after_usd = after.total_cost_usd
119
+ else:
120
+ self.estimated_output_tokens = None
121
+ self.estimated_output_cost_usd = None
122
+ self.estimated_total_cost_before_usd = None
123
+ self.estimated_total_cost_after_usd = None
97
124
  return self
98
125
 
99
126
  def summary(self) -> str:
@@ -118,6 +145,14 @@ class CompressionStats:
118
145
  "est. input cost: "
119
146
  f"${before_usd:.6f} -> ${after_usd:.6f} (saved ${saved:.6f})"
120
147
  )
148
+ out_usd = self.estimated_output_cost_usd
149
+ out_tok = self.estimated_output_tokens
150
+ if out_usd is not None and out_tok is not None:
151
+ lines.append(f"est. output cost: ${out_usd:.6f} ({out_tok} tokens)")
152
+ total_before = self.estimated_total_cost_before_usd
153
+ total_after = self.estimated_total_cost_after_usd
154
+ if total_before is not None and total_after is not None:
155
+ lines.append(f"est. total: ${total_before:.6f} -> ${total_after:.6f}")
121
156
  return "\n".join(lines)
122
157
 
123
158
  def to_dict(self) -> dict[str, Any]:
@@ -145,6 +180,10 @@ class CompressionStats:
145
180
  "estimated_input_cost_before_usd": self.estimated_input_cost_before_usd,
146
181
  "estimated_input_cost_after_usd": self.estimated_input_cost_after_usd,
147
182
  "estimated_cost_saved_usd": self.estimated_cost_saved_usd,
183
+ "estimated_output_tokens": self.estimated_output_tokens,
184
+ "estimated_output_cost_usd": self.estimated_output_cost_usd,
185
+ "estimated_total_cost_before_usd": self.estimated_total_cost_before_usd,
186
+ "estimated_total_cost_after_usd": self.estimated_total_cost_after_usd,
148
187
  }
149
188
 
150
189
 
@@ -8,6 +8,7 @@ import tiktoken
8
8
  from contextpress.models import Conversation, Turn
9
9
  from contextpress.stats import count_turn_tokens, get_encoding
10
10
  from contextpress.strategies.base import BaseStrategy
11
+ from contextpress.tools import tool_group_indices
11
12
 
12
13
 
13
14
  def _truncate_system_turn(turn: Turn, encoding: tiktoken.Encoding, max_tokens: int) -> Turn:
@@ -73,9 +74,17 @@ class BudgetStrategy(BaseStrategy):
73
74
  keep = min(2, len(ns_positions))
74
75
  protected = set(ns_positions[-keep:]) if keep else set()
75
76
  removable = [i for i in ns_positions if i not in protected]
76
- if removable:
77
- turns.pop(removable[0])
78
- n_removed += 1
77
+ group: list[int] | None = None
78
+ for idx in removable:
79
+ candidate = tool_group_indices(turns, idx)
80
+ if any(g in protected for g in candidate):
81
+ continue
82
+ group = candidate
83
+ break
84
+ if group is not None:
85
+ for g in sorted(group, reverse=True):
86
+ turns.pop(g)
87
+ n_removed += 1
79
88
  continue
80
89
 
81
90
  # Last resort: truncate system (see behavior contract note on invariant 1)
@@ -6,6 +6,7 @@ import re
6
6
  from contextpress.models import Conversation, Turn
7
7
  from contextpress.normalizer import apply_text_to_turn, extract_text_for_processing
8
8
  from contextpress.strategies.base import BaseStrategy
9
+ from contextpress.tools import has_tool_marker
9
10
 
10
11
  # Curated lists — longer phrases first for safe replacement order
11
12
  FILLER_PHRASES = [
@@ -70,16 +71,6 @@ ACKNOWLEDGEMENT_PHRASES = [
70
71
  "thank you for that",
71
72
  ]
72
73
 
73
- _TOOL_MARKERS = ("tool_calls", "tool_call", "tool_use", "tool_result", "<tool", "[tool")
74
-
75
-
76
- def _has_tool_marker(turn: Turn) -> bool:
77
- meta = turn.metadata or {}
78
- if any(k in meta for k in ("tool_calls", "tool_call", "tool_use", "tool_result")):
79
- return True
80
- text = extract_text_for_processing(turn).lower()
81
- return any(m in text for m in _TOOL_MARKERS)
82
-
83
74
 
84
75
  def _build_filler_pattern() -> re.Pattern[str]:
85
76
  pattern_parts: list[str] = []
@@ -155,7 +146,7 @@ class FillerStrategy(BaseStrategy):
155
146
  new_turns.append(nt)
156
147
  continue
157
148
 
158
- if self.conv_type == "agent" and _has_tool_marker(turn):
149
+ if has_tool_marker(turn):
159
150
  new_turns.append(copy.deepcopy(turn))
160
151
  continue
161
152
 
@@ -9,6 +9,7 @@ import re
9
9
  from contextpress.models import Conversation, Turn
10
10
  from contextpress.normalizer import apply_text_to_turn, extract_text_for_processing
11
11
  from contextpress.strategies.base import BaseStrategy
12
+ from contextpress.tools import minify_tool_fields
12
13
 
13
14
  _CODE_FENCE = re.compile(r"(```[\s\S]*?```)", re.MULTILINE)
14
15
  _MULTI_BLANK = re.compile(r"\n{3,}")
@@ -85,9 +86,10 @@ class StructureStrategy(BaseStrategy):
85
86
  text = extract_text_for_processing(turn)
86
87
  compacted = compact_structure_text(text, aggressiveness=self.aggressiveness)
87
88
  if compacted != text:
88
- new_turns.append(apply_text_to_turn(turn, compacted))
89
+ nt = apply_text_to_turn(turn, compacted)
89
90
  else:
90
- new_turns.append(copy.deepcopy(turn))
91
+ nt = copy.deepcopy(turn)
92
+ new_turns.append(minify_tool_fields(nt))
91
93
  return Conversation(
92
94
  turns=new_turns,
93
95
  type=conversation.type,