contextpress 0.6.2__tar.gz → 0.6.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {contextpress-0.6.2 → contextpress-0.6.3}/CHANGELOG.md +9 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/PKG-INFO +16 -2
- {contextpress-0.6.2 → contextpress-0.6.3}/README.md +14 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/ROADMAP.md +1 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/__init__.py +1 -1
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/core.py +17 -1
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/normalizer.py +8 -9
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/stats.py +38 -3
- contextpress-0.6.3/examples/langchain_roundtrip.py +39 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/pyproject.toml +1 -1
- contextpress-0.6.3/tests/test_v063.py +101 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/.gitignore +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/AGENTS.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/AUDIT.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/CITATION.cff +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/CONTRIBUTING.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/LICENSE +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/NOTICE +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/_bootstrap.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/compression.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/costs.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/llm/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/llm/_helpers.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/llm/adapters.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/llm/base.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/models.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/profiles.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/py.typed +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/registry.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/base.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/budget.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/filler.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/recency.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/repetition.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/resolution.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/strategies/structure.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/text_sim.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/contextpress/warnings_capture.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/agent_json_compress.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/agent_pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/benchmark_presets.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/dry_run_preview.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/estimate_and_stats.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/llm_tier_claude.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/llm_tier_gemini.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/llm_tier_ollama.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/llm_tier_openai.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/pick_preset.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/examples/structure_and_cost.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/__init__.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/01_filler_heavy.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/02_resolution_thread.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/03_repetition.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/04_long_history.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/05_agent_tools.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/06_rag_chunks.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/07_short_stable.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/08_mixed_ack_resolution.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/09_agent_tool_json.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/10_agent_repeated_logs.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/11_agent_mixed.json +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/fixtures/chats/README.md +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_budget.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_filler.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_fixture_chats.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_llm_helpers.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_models.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_normalizer.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_pipeline.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_recency.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_repetition.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_resolution.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_stats.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v03.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v04.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v05.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v051.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v052.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v053.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v054.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v056.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v058.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v060.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v061.py +0 -0
- {contextpress-0.6.2 → contextpress-0.6.3}/tests/test_v062.py +0 -0
|
@@ -4,6 +4,15 @@ All notable changes to `contextpress` are recorded here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/)
|
|
5
5
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
6
|
|
|
7
|
+
## [0.6.3] - 2026-08-12
|
|
8
|
+
|
|
9
|
+
- **LangChain round-trip** — ``compress()`` maps remaining turns back onto their original
|
|
10
|
+
message objects (not list index), so dropped turns no longer remap roles/content.
|
|
11
|
+
- **``output_tokens`` on cost stats** — ``attach_cost(output_tokens=...)``,
|
|
12
|
+
``compress(..., output_tokens=...)``, and ``ContextManager(cost_output_tokens=...)``
|
|
13
|
+
add assumed completion USD; ``summary()`` prints output + total when set.
|
|
14
|
+
- Example: `examples/langchain_roundtrip.py`.
|
|
15
|
+
|
|
7
16
|
## [0.6.2] - 2026-07-27
|
|
8
17
|
|
|
9
18
|
- **Agent-oriented fixtures** — three offline agent threads under `tests/fixtures/chats/`
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: contextpress
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.3
|
|
4
4
|
Summary: Deterministic context compression for LLM chat, RAG, and agent pipelines
|
|
5
5
|
Project-URL: Homepage, https://github.com/Taha-azizi/contextpress
|
|
6
6
|
Project-URL: Documentation, https://github.com/Taha-azizi/contextpress#readme
|
|
@@ -436,6 +436,20 @@ print(result.summary())
|
|
|
436
436
|
# est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
|
|
437
437
|
```
|
|
438
438
|
|
|
439
|
+
**Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
|
|
440
|
+
|
|
441
|
+
```python
|
|
442
|
+
cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
|
|
443
|
+
result = cm.compress(messages, token_budget=2000, return_stats=True)
|
|
444
|
+
print(result.summary())
|
|
445
|
+
# ...
|
|
446
|
+
# est. output cost: $0.000120 (200 tokens)
|
|
447
|
+
# est. total: $0.000246 -> $0.000181
|
|
448
|
+
```
|
|
449
|
+
|
|
450
|
+
LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
|
|
451
|
+
dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
|
|
452
|
+
|
|
439
453
|
See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
|
|
440
454
|
|
|
441
455
|
## Tier 1 vs Tier 2 (classical NLP vs LLM)
|
|
@@ -195,6 +195,20 @@ print(result.summary())
|
|
|
195
195
|
# est. input cost: $0.000126 -> $0.000061 (saved $0.000065) # when cost_provider set
|
|
196
196
|
```
|
|
197
197
|
|
|
198
|
+
**Assumed completion tokens** (0.6.3+, opt-in; output cost is unchanged by compression):
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai", cost_output_tokens=200)
|
|
202
|
+
result = cm.compress(messages, token_budget=2000, return_stats=True)
|
|
203
|
+
print(result.summary())
|
|
204
|
+
# ...
|
|
205
|
+
# est. output cost: $0.000120 (200 tokens)
|
|
206
|
+
# est. total: $0.000246 -> $0.000181
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
LangChain-style message objects (``.type`` / ``.content``) round-trip through ``compress()``;
|
|
210
|
+
dropped turns keep their original object types. See `examples/langchain_roundtrip.py`.
|
|
211
|
+
|
|
198
212
|
See [`ROADMAP.md`](ROADMAP.md) for positioning vs heavier compression stacks and the 0.6.x plan.
|
|
199
213
|
|
|
200
214
|
## Tier 1 vs Tier 2 (classical NLP vs LLM)
|
|
@@ -40,5 +40,6 @@ deterministic Tier‑1 NLP for chat / RAG / agent **message histories**, with op
|
|
|
40
40
|
| **0.6.0** | `structure` stage + `estimate_cost()` + this roadmap — shipped |
|
|
41
41
|
| **0.6.1** | Wire estimated USD into `CompressionStats` / reports — shipped |
|
|
42
42
|
| **0.6.2** | Agent-oriented fixtures for JSON/tool payloads; `summary()` report — shipped |
|
|
43
|
+
| **0.6.3** | LangChain compress round-trip; `output_tokens` on cost stats / `summary()` — shipped |
|
|
43
44
|
|
|
44
45
|
Stay classical-NLP-first; keep optional LLM extras optional.
|
|
@@ -48,6 +48,7 @@ class ContextManager:
|
|
|
48
48
|
llm_max_summary_tokens: int = 2048,
|
|
49
49
|
llm_mode: str = "replace_all",
|
|
50
50
|
cost_provider: str | None = None,
|
|
51
|
+
cost_output_tokens: int = 0,
|
|
51
52
|
):
|
|
52
53
|
if type not in PROFILES:
|
|
53
54
|
raise ValueError(f"unknown context type {type!r}")
|
|
@@ -65,6 +66,7 @@ class ContextManager:
|
|
|
65
66
|
self.llm_mode = llm_mode
|
|
66
67
|
# When set, compress(..., return_stats=True) attaches USD fields on stats.
|
|
67
68
|
self.cost_provider = cost_provider
|
|
69
|
+
self.cost_output_tokens = int(cost_output_tokens)
|
|
68
70
|
self._custom_stages: dict[str, StageConfig] = {}
|
|
69
71
|
|
|
70
72
|
def estimate_tokens(self, messages: Any, *, model: str | None = None) -> int:
|
|
@@ -185,6 +187,7 @@ class ContextManager:
|
|
|
185
187
|
return_stats: bool = False,
|
|
186
188
|
dry_run: bool = False,
|
|
187
189
|
cost_provider: str | None = None,
|
|
190
|
+
output_tokens: int | None = None,
|
|
188
191
|
) -> Any | CompressionResult:
|
|
189
192
|
"""Run the pipeline; return value matches input shape (dict list, tuples, strings, etc.).
|
|
190
193
|
|
|
@@ -193,6 +196,8 @@ class ContextManager:
|
|
|
193
196
|
With ``dry_run=True``, runs Tier 1 only (no LLM calls) and returns the original messages.
|
|
194
197
|
When ``cost_provider`` (or ``self.cost_provider``) is set and stats are returned,
|
|
195
198
|
``stats`` includes approximate input USD before/after compression.
|
|
199
|
+
``output_tokens`` (or ``self.cost_output_tokens``) adds an assumed completion cost
|
|
200
|
+
that is unchanged by compression.
|
|
196
201
|
"""
|
|
197
202
|
if dry_run:
|
|
198
203
|
return_stats = True
|
|
@@ -228,7 +233,14 @@ class ContextManager:
|
|
|
228
233
|
stats.warnings_emitted = captured
|
|
229
234
|
prov = cost_provider if cost_provider is not None else self.cost_provider
|
|
230
235
|
if prov is not None:
|
|
231
|
-
|
|
236
|
+
out_tok = (
|
|
237
|
+
output_tokens if output_tokens is not None else self.cost_output_tokens
|
|
238
|
+
)
|
|
239
|
+
stats.attach_cost(
|
|
240
|
+
provider=prov,
|
|
241
|
+
model=self.model or "gpt-4o-mini",
|
|
242
|
+
output_tokens=out_tok,
|
|
243
|
+
)
|
|
232
244
|
if dry_run:
|
|
233
245
|
messages_out = denormalize_output(clone_conversation(conv), ctx)
|
|
234
246
|
else:
|
|
@@ -249,6 +261,7 @@ class ContextManager:
|
|
|
249
261
|
return_stats: bool = False,
|
|
250
262
|
dry_run: bool = False,
|
|
251
263
|
cost_provider: str | None = None,
|
|
264
|
+
output_tokens: int | None = None,
|
|
252
265
|
) -> list[Any] | list[CompressionResult]:
|
|
253
266
|
"""Run ``compress()`` on each conversation in ``conversations``."""
|
|
254
267
|
if not isinstance(conversations, list):
|
|
@@ -263,6 +276,7 @@ class ContextManager:
|
|
|
263
276
|
return_stats=return_stats,
|
|
264
277
|
dry_run=dry_run,
|
|
265
278
|
cost_provider=cost_provider,
|
|
279
|
+
output_tokens=output_tokens,
|
|
266
280
|
)
|
|
267
281
|
for messages in conversations
|
|
268
282
|
]
|
|
@@ -278,6 +292,7 @@ class ContextManager:
|
|
|
278
292
|
return_stats: bool = False,
|
|
279
293
|
dry_run: bool = False,
|
|
280
294
|
cost_provider: str | None = None,
|
|
295
|
+
output_tokens: int | None = None,
|
|
281
296
|
) -> Any | CompressionResult:
|
|
282
297
|
"""Async wrapper around ``compress()`` (runs in a worker thread)."""
|
|
283
298
|
return await asyncio.to_thread(
|
|
@@ -290,6 +305,7 @@ class ContextManager:
|
|
|
290
305
|
return_stats=return_stats,
|
|
291
306
|
dry_run=dry_run,
|
|
292
307
|
cost_provider=cost_provider,
|
|
308
|
+
output_tokens=output_tokens,
|
|
293
309
|
)
|
|
294
310
|
|
|
295
311
|
def set_compression(self, compression: str) -> None:
|
|
@@ -247,15 +247,14 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
|
|
|
247
247
|
return out
|
|
248
248
|
|
|
249
249
|
if fmt == "langchain":
|
|
250
|
-
# Reconstruct LangChain objects
|
|
251
|
-
|
|
252
|
-
if not lc_objs:
|
|
253
|
-
return []
|
|
250
|
+
# Reconstruct LangChain objects from the turn's original message, not list index
|
|
251
|
+
# (dropped turns would otherwise remap remaining content onto the wrong objects).
|
|
254
252
|
result = []
|
|
255
|
-
for
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
253
|
+
for t in turns:
|
|
254
|
+
orig = t.metadata.get("_lc_original") or t.metadata.get("_lc_obj")
|
|
255
|
+
text = _turn_to_plain_text(t)
|
|
256
|
+
if orig is not None:
|
|
257
|
+
obj = copy.copy(orig)
|
|
259
258
|
if hasattr(obj, "content"):
|
|
260
259
|
try:
|
|
261
260
|
obj.content = text
|
|
@@ -270,7 +269,7 @@ def denormalize_output(conversation: Conversation, ctx: dict[str, Any]) -> Any:
|
|
|
270
269
|
self.type = role
|
|
271
270
|
self.content = content
|
|
272
271
|
|
|
273
|
-
result.append(_Msg(t.role,
|
|
272
|
+
result.append(_Msg(t.role, text))
|
|
274
273
|
return result
|
|
275
274
|
|
|
276
275
|
# dict_list
|
|
@@ -57,6 +57,11 @@ class CompressionStats:
|
|
|
57
57
|
cost_model: str | None = None
|
|
58
58
|
estimated_input_cost_before_usd: float | None = None
|
|
59
59
|
estimated_input_cost_after_usd: float | None = None
|
|
60
|
+
# Optional completion-side estimate (0.6.3+); same before/after (compression is input-only)
|
|
61
|
+
estimated_output_tokens: int | None = None
|
|
62
|
+
estimated_output_cost_usd: float | None = None
|
|
63
|
+
estimated_total_cost_before_usd: float | None = None
|
|
64
|
+
estimated_total_cost_after_usd: float | None = None
|
|
60
65
|
|
|
61
66
|
@property
|
|
62
67
|
def turns_removed(self) -> int:
|
|
@@ -86,14 +91,32 @@ class CompressionStats:
|
|
|
86
91
|
*,
|
|
87
92
|
provider: str = "openai",
|
|
88
93
|
model: str | None = "gpt-4o-mini",
|
|
94
|
+
output_tokens: int = 0,
|
|
89
95
|
) -> CompressionStats:
|
|
90
|
-
"""Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self.
|
|
91
|
-
|
|
92
|
-
|
|
96
|
+
"""Fill USD fields from ``tokens_before`` / ``tokens_after``. Returns self.
|
|
97
|
+
|
|
98
|
+
``output_tokens`` is an assumed completion size (unchanged by compression).
|
|
99
|
+
"""
|
|
100
|
+
before = estimate_token_cost(
|
|
101
|
+
self.tokens_before, provider=provider, model=model, output_tokens=output_tokens
|
|
102
|
+
)
|
|
103
|
+
after = estimate_token_cost(
|
|
104
|
+
self.tokens_after, provider=provider, model=model, output_tokens=output_tokens
|
|
105
|
+
)
|
|
93
106
|
self.cost_provider = before.provider
|
|
94
107
|
self.cost_model = before.model
|
|
95
108
|
self.estimated_input_cost_before_usd = before.input_cost_usd
|
|
96
109
|
self.estimated_input_cost_after_usd = after.input_cost_usd
|
|
110
|
+
if output_tokens > 0:
|
|
111
|
+
self.estimated_output_tokens = after.output_tokens
|
|
112
|
+
self.estimated_output_cost_usd = after.output_cost_usd
|
|
113
|
+
self.estimated_total_cost_before_usd = before.total_cost_usd
|
|
114
|
+
self.estimated_total_cost_after_usd = after.total_cost_usd
|
|
115
|
+
else:
|
|
116
|
+
self.estimated_output_tokens = None
|
|
117
|
+
self.estimated_output_cost_usd = None
|
|
118
|
+
self.estimated_total_cost_before_usd = None
|
|
119
|
+
self.estimated_total_cost_after_usd = None
|
|
97
120
|
return self
|
|
98
121
|
|
|
99
122
|
def summary(self) -> str:
|
|
@@ -118,6 +141,14 @@ class CompressionStats:
|
|
|
118
141
|
"est. input cost: "
|
|
119
142
|
f"${before_usd:.6f} -> ${after_usd:.6f} (saved ${saved:.6f})"
|
|
120
143
|
)
|
|
144
|
+
out_usd = self.estimated_output_cost_usd
|
|
145
|
+
out_tok = self.estimated_output_tokens
|
|
146
|
+
if out_usd is not None and out_tok is not None:
|
|
147
|
+
lines.append(f"est. output cost: ${out_usd:.6f} ({out_tok} tokens)")
|
|
148
|
+
total_before = self.estimated_total_cost_before_usd
|
|
149
|
+
total_after = self.estimated_total_cost_after_usd
|
|
150
|
+
if total_before is not None and total_after is not None:
|
|
151
|
+
lines.append(f"est. total: ${total_before:.6f} -> ${total_after:.6f}")
|
|
121
152
|
return "\n".join(lines)
|
|
122
153
|
|
|
123
154
|
def to_dict(self) -> dict[str, Any]:
|
|
@@ -145,6 +176,10 @@ class CompressionStats:
|
|
|
145
176
|
"estimated_input_cost_before_usd": self.estimated_input_cost_before_usd,
|
|
146
177
|
"estimated_input_cost_after_usd": self.estimated_input_cost_after_usd,
|
|
147
178
|
"estimated_cost_saved_usd": self.estimated_cost_saved_usd,
|
|
179
|
+
"estimated_output_tokens": self.estimated_output_tokens,
|
|
180
|
+
"estimated_output_cost_usd": self.estimated_output_cost_usd,
|
|
181
|
+
"estimated_total_cost_before_usd": self.estimated_total_cost_before_usd,
|
|
182
|
+
"estimated_total_cost_after_usd": self.estimated_total_cost_after_usd,
|
|
148
183
|
}
|
|
149
184
|
|
|
150
185
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""LangChain-style objects round-trip through compress() (0.6.3+).
|
|
2
|
+
|
|
3
|
+
Uses duck-typed message objects (``.type`` / ``.content``), so no LangChain
|
|
4
|
+
install is required. Real ``HumanMessage`` / ``AIMessage`` lists work the same way.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from contextpress import ContextManager
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Msg:
|
|
13
|
+
def __init__(self, typ: str, content: str):
|
|
14
|
+
self.type = typ
|
|
15
|
+
self.content = content
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
messages = [
|
|
19
|
+
Msg("system", "You are a concise assistant."),
|
|
20
|
+
Msg("human", "We've decided on using the new pipeline. Basically just confirm."),
|
|
21
|
+
Msg("ai", "Sounds good"),
|
|
22
|
+
Msg("human", "Schedule the api-v2 staging deploy for Monday."),
|
|
23
|
+
Msg("ai", "Confirmed. Monday staging deploy is scheduled."),
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
cm = ContextManager(
|
|
27
|
+
type="chat",
|
|
28
|
+
model="gpt-4o-mini",
|
|
29
|
+
compression="high",
|
|
30
|
+
cost_provider="openai",
|
|
31
|
+
cost_output_tokens=150,
|
|
32
|
+
)
|
|
33
|
+
result = cm.compress(messages, token_budget=None, return_stats=True)
|
|
34
|
+
|
|
35
|
+
print(result.summary())
|
|
36
|
+
print()
|
|
37
|
+
for m in result.messages:
|
|
38
|
+
preview = str(m.content)[:80]
|
|
39
|
+
print(f"{m.type}: {preview}")
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""0.6.3 — LangChain compress round-trip and output_tokens on cost stats."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from contextpress import ContextManager
|
|
6
|
+
from contextpress.stats import CompressionStats
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class _FakeMsg:
|
|
10
|
+
def __init__(self, typ: str, content: str):
|
|
11
|
+
self.type = typ
|
|
12
|
+
self.content = content
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_langchain_compress_roundtrip_keeps_object_shape():
|
|
16
|
+
msgs = [
|
|
17
|
+
_FakeMsg("system", "You are a concise assistant."),
|
|
18
|
+
_FakeMsg("human", "Summarize the deploy status for api-v2."),
|
|
19
|
+
_FakeMsg("ai", "Checking logs now."),
|
|
20
|
+
]
|
|
21
|
+
cm = ContextManager(type="chat", compression="low")
|
|
22
|
+
out = cm.compress(msgs, token_budget=None)
|
|
23
|
+
assert isinstance(out, list)
|
|
24
|
+
assert len(out) >= 1
|
|
25
|
+
assert all(hasattr(m, "content") for m in out)
|
|
26
|
+
assert out[0].type == "system"
|
|
27
|
+
assert out[0].content == "You are a concise assistant."
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_langchain_dropped_turn_does_not_remap_roles():
|
|
31
|
+
"""Index-based denormalize would put later content onto the dropped object's type."""
|
|
32
|
+
msgs = [
|
|
33
|
+
_FakeMsg("system", "sys"),
|
|
34
|
+
_FakeMsg("human", "We've decided on using Monday."),
|
|
35
|
+
_FakeMsg("ai", "Sounds good"),
|
|
36
|
+
_FakeMsg("human", "Please confirm the Monday plan in detail."),
|
|
37
|
+
_FakeMsg("ai", "Confirmed. Monday it is, with the new pipeline."),
|
|
38
|
+
]
|
|
39
|
+
original = [(m.type, m.content) for m in msgs]
|
|
40
|
+
cm = ContextManager(type="chat", compression="high")
|
|
41
|
+
out = cm.compress(msgs, token_budget=None)
|
|
42
|
+
assert [(m.type, m.content) for m in msgs] == original
|
|
43
|
+
types = [getattr(m, "type", None) for m in out]
|
|
44
|
+
assert types[0] == "system"
|
|
45
|
+
for m in out:
|
|
46
|
+
if getattr(m, "type", None) == "human":
|
|
47
|
+
assert "Sounds good" not in str(m.content)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_attach_cost_output_tokens():
|
|
51
|
+
stats = CompressionStats(tokens_before=1_000_000, tokens_after=500_000)
|
|
52
|
+
stats.attach_cost(provider="openai", model="gpt-4o-mini", output_tokens=100_000)
|
|
53
|
+
assert stats.estimated_input_cost_before_usd == 0.15
|
|
54
|
+
assert stats.estimated_input_cost_after_usd == 0.075
|
|
55
|
+
assert stats.estimated_output_tokens == 100_000
|
|
56
|
+
assert stats.estimated_output_cost_usd == 0.06
|
|
57
|
+
assert stats.estimated_total_cost_before_usd == 0.21
|
|
58
|
+
assert stats.estimated_total_cost_after_usd == 0.135
|
|
59
|
+
text = stats.summary()
|
|
60
|
+
assert "est. output cost:" in text
|
|
61
|
+
assert "100000 tokens" in text
|
|
62
|
+
assert "est. total:" in text
|
|
63
|
+
d = stats.to_dict()
|
|
64
|
+
assert d["estimated_output_cost_usd"] == 0.06
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def test_attach_cost_zero_output_leaves_output_fields_none():
|
|
68
|
+
stats = CompressionStats(tokens_before=1000, tokens_after=500)
|
|
69
|
+
stats.attach_cost(provider="openai", model="gpt-4o-mini", output_tokens=0)
|
|
70
|
+
assert stats.estimated_output_tokens is None
|
|
71
|
+
assert stats.estimated_output_cost_usd is None
|
|
72
|
+
assert "est. output cost" not in stats.summary()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_compress_output_tokens_kwarg():
|
|
76
|
+
cm = ContextManager(type="chat", model="gpt-4o-mini", cost_provider="openai")
|
|
77
|
+
result = cm.compress(
|
|
78
|
+
[{"role": "user", "content": "hello " * 40}],
|
|
79
|
+
token_budget=None,
|
|
80
|
+
return_stats=True,
|
|
81
|
+
output_tokens=200,
|
|
82
|
+
)
|
|
83
|
+
assert result.stats.estimated_output_tokens == 200
|
|
84
|
+
assert result.stats.estimated_output_cost_usd is not None
|
|
85
|
+
assert result.stats.estimated_total_cost_after_usd is not None
|
|
86
|
+
assert "est. total:" in result.summary()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_cost_output_tokens_constructor_default():
|
|
90
|
+
cm = ContextManager(
|
|
91
|
+
type="chat",
|
|
92
|
+
model="gpt-4o-mini",
|
|
93
|
+
cost_provider="openai",
|
|
94
|
+
cost_output_tokens=50,
|
|
95
|
+
)
|
|
96
|
+
result = cm.compress(
|
|
97
|
+
[{"role": "user", "content": "hello " * 20}],
|
|
98
|
+
token_budget=None,
|
|
99
|
+
return_stats=True,
|
|
100
|
+
)
|
|
101
|
+
assert result.stats.estimated_output_tokens == 50
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|