tokenmizer 0.2.6__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/CHANGELOG.md +15 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/PKG-INFO +57 -14
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/README.md +56 -13
- tokenmizer-0.3.0/docs/assets/demo.gif +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/pyproject.toml +1 -1
- tokenmizer-0.3.0/scripts/gen_demo_gif.py +108 -0
- tokenmizer-0.3.0/server.json +34 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/test_api_endpoint.py +34 -1
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/__init__.py +1 -1
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/app.py +86 -13
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/providers/providers.py +108 -0
- tokenmizer-0.2.6/server.json +0 -29
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/marketplace.json +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/plugin.json +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/analyze/SKILL.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/checkpoint/SKILL.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/resume/SKILL.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/stats/SKILL.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/ISSUE_TEMPLATE/extraction_miss.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/workflows/ci.yml +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/workflows/release.yml +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.gitignore +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.mcp.json +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/CONTRIBUTING.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/Dockerfile +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/LICENSE +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/SECURITY.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/TESTING.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/USAGE.md +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner_v2.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner_v3.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/graph_retrieval/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/graph_retrieval/runner.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/latency/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/latency/runner.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/resume_quality/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docker-compose.yml +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docs/assets/architecture.svg +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docs/assets/logo.svg +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/examples/basic_usage.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/install.sh +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/mcp_e2e_check.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/run_stdlib_tests.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/setup.sh +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/static_audit.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/chaos/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/chaos/test_recovery.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/conftest.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/test_checkpoint.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/memory_accuracy/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/memory_accuracy/test_retention.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_cache.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_compression_correctness.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_decision_cache_async.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_file_intelligence.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_graph.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_graph_persistence.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_hybrid_extractor.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_rate_limiter.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_security.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_tokenizer.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_validator.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/agents/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/analytics/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/analytics/engine.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/rate_limiter.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/checkpoints/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/checkpoints/manager.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/cli.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/engine.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/output_trimmer.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/window.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/config/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/config/settings.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/dto.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/errors.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/tokenizer.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/dashboard/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/dashboard/page.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/filters/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/filters/file_intelligence.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/decision_tracker.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/graph.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/helpers.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/hybrid_extractor.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/types.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/validator.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/visualization.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/mcp/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/mcp/server.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/providers/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/auth.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/middleware.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/redaction.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/semantic_cache/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/semantic_cache/cache.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/state/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/state/backend.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/storage/__init__.py +0 -0
- {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer.yaml +0 -0
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.3.0] — 2026-07-02 — true SSE streaming
|
|
4
|
+
|
|
5
|
+
- **`stream: true` now works** — real passthrough streaming in OpenAI
|
|
6
|
+
`chat.completion.chunk` SSE format for Anthropic, OpenAI, DeepSeek,
|
|
7
|
+
Mistral, OpenRouter, Grok (OpenAI-compatible base) and Ollama. Cursor,
|
|
8
|
+
Continue.dev and every streaming client can now point at TokenMizer
|
|
9
|
+
without config changes.
|
|
10
|
+
- All input-side layers (file intelligence, compression, graph memory,
|
|
11
|
+
context injection) apply to streamed requests; output trimming is
|
|
12
|
+
skipped in stream mode by design. Cache hits stream as a single chunk;
|
|
13
|
+
post-stream analytics + cache writes preserved.
|
|
14
|
+
- Providers without passthrough support return an explicit 501 (never a
|
|
15
|
+
fake buffered stream). Mid-stream provider failures emit an SSE `error`
|
|
16
|
+
event instead of silently truncating.
|
|
17
|
+
|
|
3
18
|
## [0.2.6] — 2026-07-02 — MCP registry readiness
|
|
4
19
|
|
|
5
20
|
- NEW console script `tokenmizer-mcp` — runs the MCP stdio server directly
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenmizer
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
6
6
|
Project-URL: Repository, https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
@@ -87,12 +87,24 @@ Description-Content-Type: text/markdown
|
|
|
87
87
|
|
|
88
88
|
<p>
|
|
89
89
|
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/v/tokenmizer?color=7c6af7&style=flat-square" alt="PyPI"/></a>
|
|
90
|
-
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/
|
|
90
|
+
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/dm/tokenmizer?color=5ee7c8&style=flat-square" alt="Downloads"/></a>
|
|
91
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80" alt="CI"/></a>
|
|
92
|
+
<a href="https://registry.modelcontextprotocol.io/v0/servers?search=tokenmizer"><img src="https://img.shields.io/badge/MCP%20Registry-published-5ee7c8?style=flat-square" alt="MCP Registry"/></a>
|
|
91
93
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-4ade80?style=flat-square"/></a>
|
|
92
|
-
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/
|
|
93
|
-
<img src="https://img.shields.io/badge/Claude%20Code-Plugin-7c6af7?style=flat-square&logo=anthropic"/>
|
|
94
|
-
<img src="https://img.shields.io/badge/MCP-Compatible-5ee7c8?style=flat-square"/>
|
|
94
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/stargazers"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=flat-square&color=f9d84a" alt="Stars"/></a>
|
|
95
95
|
</p>
|
|
96
|
+
|
|
97
|
+
<p>
|
|
98
|
+
<a href="#quick-start"><b>Quick Start</b></a> ·
|
|
99
|
+
<a href="#how-tokenmizer-solves-it"><b>How it works</b></a> ·
|
|
100
|
+
<a href="#benchmarks"><b>Benchmarks</b></a> ·
|
|
101
|
+
<a href="#claude-code-integration"><b>Claude Code</b></a> ·
|
|
102
|
+
<a href="#contributing"><b>Contributing</b></a>
|
|
103
|
+
</p>
|
|
104
|
+
|
|
105
|
+
<img src="docs/assets/demo.gif" width="860" alt="TokenMizer demo: 40-turn session checkpointed at 87% context, resumed next day in 233 tokens"/>
|
|
106
|
+
<br/>
|
|
107
|
+
<sub>Real run: 25-node graph, checkpoint <code>ckpt_21a0959c3ddf</code>, 233-token resume. Regenerate with <code>python scripts/gen_demo_gif.py</code>.</sub>
|
|
96
108
|
</div>
|
|
97
109
|
|
|
98
110
|
---
|
|
@@ -193,13 +205,9 @@ response = client.chat.completions.create(
|
|
|
193
205
|
)
|
|
194
206
|
```
|
|
195
207
|
|
|
196
|
-
>
|
|
197
|
-
>
|
|
198
|
-
>
|
|
199
|
-
>
|
|
200
|
-
> **Cursor:** Settings → Models → your model entry → disable "Stream responses"
|
|
201
|
-
> **Continue.dev:** in `config.json`, set `"streamResponses": false` for the
|
|
202
|
-
> TokenMizer provider entry
|
|
208
|
+
> ✅ **Streaming works** (v0.3+): `stream: true` gives real SSE passthrough for
|
|
209
|
+
> Anthropic, OpenAI, DeepSeek, Mistral, OpenRouter, Grok and Ollama. Cursor and
|
|
210
|
+
> Continue.dev work with default settings — no config changes needed.
|
|
203
211
|
|
|
204
212
|
---
|
|
205
213
|
|
|
@@ -505,16 +513,49 @@ tokenmizer stats
|
|
|
505
513
|
|
|
506
514
|
---
|
|
507
515
|
|
|
516
|
+
## Roadmap
|
|
517
|
+
|
|
518
|
+
| Version | Focus |
|
|
519
|
+
|---|---|
|
|
520
|
+
| **v0.3** | SSE streaming passthrough (checkpoint on stream close) |
|
|
521
|
+
| v0.4 | Cross-session memory · embedding-based edge linking |
|
|
522
|
+
| v0.5 | Per-node storage schema (scale past 200-node graphs) |
|
|
523
|
+
| Research | Real-transcript benchmark suite → paper ([tokenmizer-research](https://github.com/Shweta-Mishra-ai/tokenmizer-research)) |
|
|
524
|
+
|
|
525
|
+
Have a use case that doesn't fit? [Open an issue](https://github.com/Shweta-Mishra-ai/tokenmizer/issues/new/choose) — extraction misses have their own issue template.
|
|
526
|
+
|
|
527
|
+
---
|
|
528
|
+
|
|
508
529
|
## Contributing
|
|
509
530
|
|
|
510
|
-
|
|
531
|
+
Contributions welcome — this project merges fast (median PR review < 1 day).
|
|
511
532
|
|
|
512
533
|
```bash
|
|
513
534
|
git clone https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
535
|
+
cd tokenmizer
|
|
514
536
|
pip install -e ".[dev]"
|
|
515
|
-
pytest tests/ -v && ruff check tokenmizer/
|
|
537
|
+
pytest tests/ -v && ruff check tokenmizer/ # 218 tests, must stay green
|
|
538
|
+
python scripts/mcp_e2e_check.py # full-pipeline e2e check
|
|
516
539
|
```
|
|
517
540
|
|
|
541
|
+
**Highest-impact areas right now:**
|
|
542
|
+
|
|
543
|
+
1. **Graph extraction quality** — real-world transcripts where extraction misses tasks/decisions (file an [extraction-miss issue](.github/ISSUE_TEMPLATE/extraction_miss.md) even if you don't fix it — the failing transcript itself is the contribution)
|
|
544
|
+
2. **SSE streaming** (v0.3 headline feature)
|
|
545
|
+
3. **Benchmark sessions** — add a real session + ground truth to `benchmarks/`
|
|
546
|
+
|
|
547
|
+
Every PR runs the full CI gauntlet (tests × 3 Python versions, lint, Docker build). See [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines and [TESTING.md](TESTING.md) for the test architecture.
|
|
548
|
+
|
|
549
|
+
---
|
|
550
|
+
|
|
551
|
+
## Support the project
|
|
552
|
+
|
|
553
|
+
TokenMizer is built and maintained by one person. If it saved you tokens, time, or a lost session:
|
|
554
|
+
|
|
555
|
+
- ⭐ **[Star the repo](https://github.com/Shweta-Mishra-ai/tokenmizer)** — the single best way to help others find it
|
|
556
|
+
- 🐛 [Report a bug](https://github.com/Shweta-Mishra-ai/tokenmizer/issues) — especially extraction misses
|
|
557
|
+
- 📣 Share your before/after token numbers (`tokenmizer stats`) — real usage data shapes the roadmap
|
|
558
|
+
|
|
518
559
|
---
|
|
519
560
|
|
|
520
561
|
## License
|
|
@@ -525,4 +566,6 @@ MIT © [Shweta Mishra](https://github.com/Shweta-Mishra-ai)
|
|
|
525
566
|
|
|
526
567
|
<div align="center">
|
|
527
568
|
<sub>Built for developers who spend too much time re-explaining their projects to AI.</sub>
|
|
569
|
+
<br/><br/>
|
|
570
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=social" alt="GitHub stars"/></a>
|
|
528
571
|
</div>
|
|
@@ -12,12 +12,24 @@
|
|
|
12
12
|
|
|
13
13
|
<p>
|
|
14
14
|
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/v/tokenmizer?color=7c6af7&style=flat-square" alt="PyPI"/></a>
|
|
15
|
-
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/
|
|
15
|
+
<a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/dm/tokenmizer?color=5ee7c8&style=flat-square" alt="Downloads"/></a>
|
|
16
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80" alt="CI"/></a>
|
|
17
|
+
<a href="https://registry.modelcontextprotocol.io/v0/servers?search=tokenmizer"><img src="https://img.shields.io/badge/MCP%20Registry-published-5ee7c8?style=flat-square" alt="MCP Registry"/></a>
|
|
16
18
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-4ade80?style=flat-square"/></a>
|
|
17
|
-
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/
|
|
18
|
-
<img src="https://img.shields.io/badge/Claude%20Code-Plugin-7c6af7?style=flat-square&logo=anthropic"/>
|
|
19
|
-
<img src="https://img.shields.io/badge/MCP-Compatible-5ee7c8?style=flat-square"/>
|
|
19
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer/stargazers"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=flat-square&color=f9d84a" alt="Stars"/></a>
|
|
20
20
|
</p>
|
|
21
|
+
|
|
22
|
+
<p>
|
|
23
|
+
<a href="#quick-start"><b>Quick Start</b></a> ·
|
|
24
|
+
<a href="#how-tokenmizer-solves-it"><b>How it works</b></a> ·
|
|
25
|
+
<a href="#benchmarks"><b>Benchmarks</b></a> ·
|
|
26
|
+
<a href="#claude-code-integration"><b>Claude Code</b></a> ·
|
|
27
|
+
<a href="#contributing"><b>Contributing</b></a>
|
|
28
|
+
</p>
|
|
29
|
+
|
|
30
|
+
<img src="docs/assets/demo.gif" width="860" alt="TokenMizer demo: 40-turn session checkpointed at 87% context, resumed next day in 233 tokens"/>
|
|
31
|
+
<br/>
|
|
32
|
+
<sub>Real run: 25-node graph, checkpoint <code>ckpt_21a0959c3ddf</code>, 233-token resume. Regenerate with <code>python scripts/gen_demo_gif.py</code>.</sub>
|
|
21
33
|
</div>
|
|
22
34
|
|
|
23
35
|
---
|
|
@@ -118,13 +130,9 @@ response = client.chat.completions.create(
|
|
|
118
130
|
)
|
|
119
131
|
```
|
|
120
132
|
|
|
121
|
-
>
|
|
122
|
-
>
|
|
123
|
-
>
|
|
124
|
-
>
|
|
125
|
-
> **Cursor:** Settings → Models → your model entry → disable "Stream responses"
|
|
126
|
-
> **Continue.dev:** in `config.json`, set `"streamResponses": false` for the
|
|
127
|
-
> TokenMizer provider entry
|
|
133
|
+
> ✅ **Streaming works** (v0.3+): `stream: true` gives real SSE passthrough for
|
|
134
|
+
> Anthropic, OpenAI, DeepSeek, Mistral, OpenRouter, Grok and Ollama. Cursor and
|
|
135
|
+
> Continue.dev work with default settings — no config changes needed.
|
|
128
136
|
|
|
129
137
|
---
|
|
130
138
|
|
|
@@ -430,16 +438,49 @@ tokenmizer stats
|
|
|
430
438
|
|
|
431
439
|
---
|
|
432
440
|
|
|
441
|
+
## Roadmap
|
|
442
|
+
|
|
443
|
+
| Version | Focus |
|
|
444
|
+
|---|---|
|
|
445
|
+
| **v0.3** | SSE streaming passthrough (checkpoint on stream close) |
|
|
446
|
+
| v0.4 | Cross-session memory · embedding-based edge linking |
|
|
447
|
+
| v0.5 | Per-node storage schema (scale past 200-node graphs) |
|
|
448
|
+
| Research | Real-transcript benchmark suite → paper ([tokenmizer-research](https://github.com/Shweta-Mishra-ai/tokenmizer-research)) |
|
|
449
|
+
|
|
450
|
+
Have a use case that doesn't fit? [Open an issue](https://github.com/Shweta-Mishra-ai/tokenmizer/issues/new/choose) — extraction misses have their own issue template.
|
|
451
|
+
|
|
452
|
+
---
|
|
453
|
+
|
|
433
454
|
## Contributing
|
|
434
455
|
|
|
435
|
-
|
|
456
|
+
Contributions welcome — this project merges fast (median PR review < 1 day).
|
|
436
457
|
|
|
437
458
|
```bash
|
|
438
459
|
git clone https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
460
|
+
cd tokenmizer
|
|
439
461
|
pip install -e ".[dev]"
|
|
440
|
-
pytest tests/ -v && ruff check tokenmizer/
|
|
462
|
+
pytest tests/ -v && ruff check tokenmizer/ # 218 tests, must stay green
|
|
463
|
+
python scripts/mcp_e2e_check.py # full-pipeline e2e check
|
|
441
464
|
```
|
|
442
465
|
|
|
466
|
+
**Highest-impact areas right now:**
|
|
467
|
+
|
|
468
|
+
1. **Graph extraction quality** — real-world transcripts where extraction misses tasks/decisions (file an [extraction-miss issue](.github/ISSUE_TEMPLATE/extraction_miss.md) even if you don't fix it — the failing transcript itself is the contribution)
|
|
469
|
+
2. **SSE streaming** (v0.3 headline feature)
|
|
470
|
+
3. **Benchmark sessions** — add a real session + ground truth to `benchmarks/`
|
|
471
|
+
|
|
472
|
+
Every PR runs the full CI gauntlet (tests × 3 Python versions, lint, Docker build). See [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines and [TESTING.md](TESTING.md) for the test architecture.
|
|
473
|
+
|
|
474
|
+
---
|
|
475
|
+
|
|
476
|
+
## Support the project
|
|
477
|
+
|
|
478
|
+
TokenMizer is built and maintained by one person. If it saved you tokens, time, or a lost session:
|
|
479
|
+
|
|
480
|
+
- ⭐ **[Star the repo](https://github.com/Shweta-Mishra-ai/tokenmizer)** — the single best way to help others find it
|
|
481
|
+
- 🐛 [Report a bug](https://github.com/Shweta-Mishra-ai/tokenmizer/issues) — especially extraction misses
|
|
482
|
+
- 📣 Share your before/after token numbers (`tokenmizer stats`) — real usage data shapes the roadmap
|
|
483
|
+
|
|
443
484
|
---
|
|
444
485
|
|
|
445
486
|
## License
|
|
@@ -450,4 +491,6 @@ MIT © [Shweta Mishra](https://github.com/Shweta-Mishra-ai)
|
|
|
450
491
|
|
|
451
492
|
<div align="center">
|
|
452
493
|
<sub>Built for developers who spend too much time re-explaining their projects to AI.</sub>
|
|
494
|
+
<br/><br/>
|
|
495
|
+
<a href="https://github.com/Shweta-Mishra-ai/tokenmizer"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=social" alt="GitHub stars"/></a>
|
|
453
496
|
</div>
|
|
Binary file
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""Generate docs/assets/demo.gif — terminal-style demo.
|
|
3
|
+
|
|
4
|
+
All numbers/outputs shown are REAL, captured from an actual run of
|
|
5
|
+
GraphMemory + CheckpointManager on the 40-turn-scale demo session
|
|
6
|
+
(checkpoint ckpt_21a0959c3ddf, 25 nodes, 233-token resume).
|
|
7
|
+
Run: python scripts/gen_demo_gif.py [output_path]
|
|
8
|
+
"""
|
|
9
|
+
import os
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from PIL import Image, ImageDraw, ImageFont
|
|
13
|
+
|
|
14
|
+
OUT = sys.argv[1] if len(sys.argv) > 1 else "docs/assets/demo.gif"
|
|
15
|
+
W, H = 920, 500
|
|
16
|
+
BG, FG = (13, 17, 23), (201, 209, 217)
|
|
17
|
+
GREEN, PURPLE, YELLOW = (63, 185, 80), (137, 87, 229), (210, 153, 34)
|
|
18
|
+
DIM, CYAN = (110, 118, 129), (57, 197, 187)
|
|
19
|
+
|
|
20
|
+
font = ImageFont.truetype("consola.ttf", 17)
|
|
21
|
+
bold = ImageFont.truetype("consolab.ttf", 17)
|
|
22
|
+
big = ImageFont.truetype("consolab.ttf", 26)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def frame(lines, title="tokenmizer - demo"):
|
|
26
|
+
im = Image.new("RGB", (W, H), BG)
|
|
27
|
+
d = ImageDraw.Draw(im)
|
|
28
|
+
d.rounded_rectangle([8, 8, W - 8, H - 8], 10, outline=(48, 54, 61), width=2)
|
|
29
|
+
for i, c in enumerate([(255, 95, 86), (255, 189, 46), (39, 201, 63)]):
|
|
30
|
+
d.ellipse([24 + i * 24, 20, 38 + i * 24, 34], fill=c)
|
|
31
|
+
d.text((W // 2 - 100, 18), title, font=font, fill=DIM)
|
|
32
|
+
y = 56
|
|
33
|
+
for txt, col, f in lines:
|
|
34
|
+
d.text((28, y), txt, font=f, fill=col)
|
|
35
|
+
y += 24
|
|
36
|
+
return im
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
frames, durs = [], []
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def add(lines, ms):
|
|
43
|
+
frames.append(frame(lines))
|
|
44
|
+
durs.append(ms)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
add([("$ pip install tokenmizer", FG, bold),
|
|
48
|
+
("Successfully installed tokenmizer-0.2.6", GREEN, font),
|
|
49
|
+
("", FG, font),
|
|
50
|
+
("$ export TOKENMIZER_ANTHROPIC_API_KEY=sk-ant-...", FG, bold),
|
|
51
|
+
("$ tokenmizer serve", FG, bold),
|
|
52
|
+
(" Proxy: http://localhost:8000/v1/chat/completions", CYAN, font),
|
|
53
|
+
(" Dashboard: http://localhost:8000", CYAN, font),
|
|
54
|
+
("", FG, font),
|
|
55
|
+
("# point any OpenAI-compatible client at localhost:8000 -", DIM, font),
|
|
56
|
+
("# one line changed, zero code rewritten", DIM, font)], 2600)
|
|
57
|
+
|
|
58
|
+
add([("session: fastapi-auth model: claude-fable-5", DIM, font),
|
|
59
|
+
("", FG, font),
|
|
60
|
+
("you > Let's build a FastAPI auth service with JWT + PostgreSQL", FG, font),
|
|
61
|
+
("ai > Decided: PostgreSQL (concurrent writes). Files: api/auth.py ...", DIM, font),
|
|
62
|
+
("you > Use bcrypt and Python 3.12", FG, font),
|
|
63
|
+
("ai > Decided: bcrypt (industry standard) ...", DIM, font),
|
|
64
|
+
("you > Login returns 422, fix it", FG, font),
|
|
65
|
+
("ai > Fixed: missing email validation in LoginRequest ...", DIM, font),
|
|
66
|
+
("you > Add refresh tokens with Redis", FG, font),
|
|
67
|
+
("ai > Decided: Redis for refresh tokens (faster revocation) ...", DIM, font),
|
|
68
|
+
(" ... 40 turns later ...", YELLOW, bold),
|
|
69
|
+
("", FG, font),
|
|
70
|
+
(" context window: 87% full", YELLOW, bold)], 3200)
|
|
71
|
+
|
|
72
|
+
add([(" context window: 87% full", YELLOW, bold),
|
|
73
|
+
("", FG, font),
|
|
74
|
+
(" * auto-checkpoint triggered *", PURPLE, bold),
|
|
75
|
+
("", FG, font),
|
|
76
|
+
(" graph: 25 nodes extracted (tasks / decisions / files / errors)", FG, font),
|
|
77
|
+
(" checkpoint: ckpt_21a0959c3ddf saved to SQLite", GREEN, font),
|
|
78
|
+
(" resume size: 233 tokens", GREEN, bold),
|
|
79
|
+
("", FG, font),
|
|
80
|
+
(" session history: ~5,800 tokens -> 233 tokens", CYAN, bold)], 3000)
|
|
81
|
+
|
|
82
|
+
add([(" -- next day, fresh session --", DIM, font),
|
|
83
|
+
("", FG, font),
|
|
84
|
+
("$ tokenmizer resume fastapi-auth", FG, bold),
|
|
85
|
+
("", FG, font),
|
|
86
|
+
("Goal: a FastAPI auth service with JWT and PostgreSQL", GREEN, font),
|
|
87
|
+
("Working on: rate limiting with slowapi", FG, font),
|
|
88
|
+
("Done: tests/test_auth.py (12 tests passing) | 422 login fix", FG, font),
|
|
89
|
+
("Decided: PostgreSQL | bcrypt | Python 3.12 |", FG, font),
|
|
90
|
+
(" Redis for refresh tokens (faster revocation)", FG, font),
|
|
91
|
+
("Changes: 'Use Redis' -> 'Redis for refresh token storage'", YELLOW, font),
|
|
92
|
+
("Continue from: Write the tests", CYAN, font),
|
|
93
|
+
("", FG, font),
|
|
94
|
+
(" [233 tokens - paste into any new session and keep going]", DIM, font)], 4200)
|
|
95
|
+
|
|
96
|
+
im = Image.new("RGB", (W, H), BG)
|
|
97
|
+
d = ImageDraw.Draw(im)
|
|
98
|
+
d.rounded_rectangle([8, 8, W - 8, H - 8], 10, outline=(48, 54, 61), width=2)
|
|
99
|
+
d.text((W // 2 - 260, H // 2 - 70), "never re-explain your project again", font=big, fill=FG)
|
|
100
|
+
d.text((W // 2 - 215, H // 2 - 20), "~5,800 tokens -> 233 tokens resume", font=big, fill=GREEN)
|
|
101
|
+
d.text((W // 2 - 130, H // 2 + 40), "pip install tokenmizer", font=bold, fill=PURPLE)
|
|
102
|
+
frames.append(im)
|
|
103
|
+
durs.append(3500)
|
|
104
|
+
|
|
105
|
+
os.makedirs(os.path.dirname(OUT), exist_ok=True)
|
|
106
|
+
frames[0].save(OUT, save_all=True, append_images=frames[1:],
|
|
107
|
+
duration=durs, loop=0, optimize=True)
|
|
108
|
+
print("demo.gif:", os.path.getsize(OUT) // 1024, "KB,", len(frames), "frames")
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
|
|
3
|
+
"name": "io.github.Shweta-Mishra-ai/tokenmizer",
|
|
4
|
+
"description": "Graph-backed session memory for LLMs: checkpoint, resume, file analysis, token-savings stats.",
|
|
5
|
+
"repository": {
|
|
6
|
+
"url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
|
|
7
|
+
"source": "github"
|
|
8
|
+
},
|
|
9
|
+
"version": "0.2.6",
|
|
10
|
+
"packages": [
|
|
11
|
+
{
|
|
12
|
+
"registryType": "pypi",
|
|
13
|
+
"identifier": "tokenmizer",
|
|
14
|
+
"version": "0.2.6",
|
|
15
|
+
"transport": { "type": "stdio" },
|
|
16
|
+
"environmentVariables": [
|
|
17
|
+
{
|
|
18
|
+
"name": "TOKENMIZER_URL",
|
|
19
|
+
"description": "Base URL of the running TokenMizer proxy (default http://localhost:8000)",
|
|
20
|
+
"isRequired": false,
|
|
21
|
+
"format": "string",
|
|
22
|
+
"isSecret": false
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"name": "TOKENMIZER_API_KEY",
|
|
26
|
+
"description": "API key if the proxy has auth enabled",
|
|
27
|
+
"isRequired": false,
|
|
28
|
+
"format": "string",
|
|
29
|
+
"isSecret": true
|
|
30
|
+
}
|
|
31
|
+
]
|
|
32
|
+
}
|
|
33
|
+
]
|
|
34
|
+
}
|
|
@@ -126,7 +126,40 @@ class TestChatCompletionsE2E:
|
|
|
126
126
|
assert kwargs.get("temperature") == 0.0
|
|
127
127
|
assert kwargs.get("top_p") == 0.9
|
|
128
128
|
|
|
129
|
-
def
|
|
129
|
+
def test_streaming_returns_sse_chunks(self, client):
|
|
130
|
+
"""v0.3: stream=true must return true SSE passthrough in OpenAI
|
|
131
|
+
chat.completion.chunk format, ending with data: [DONE]."""
|
|
132
|
+
c, fake = client
|
|
133
|
+
|
|
134
|
+
async def fake_stream(**kwargs):
|
|
135
|
+
for piece in ("Hello", " streamed", " world"):
|
|
136
|
+
yield piece
|
|
137
|
+
|
|
138
|
+
fake.chat_stream = fake_stream
|
|
139
|
+
r = c.post("/v1/chat/completions", json={
|
|
140
|
+
"messages": [{"role": "user", "content": "stream me a story please"}],
|
|
141
|
+
"stream": True,
|
|
142
|
+
})
|
|
143
|
+
assert r.status_code == 200, r.text
|
|
144
|
+
assert "text/event-stream" in r.headers["content-type"]
|
|
145
|
+
body = r.text
|
|
146
|
+
assert "chat.completion.chunk" in body
|
|
147
|
+
assert '"content": "Hello"' in body
|
|
148
|
+
assert '"content": " streamed"' in body
|
|
149
|
+
assert '"finish_reason": "stop"' in body
|
|
150
|
+
assert body.rstrip().endswith("data: [DONE]")
|
|
151
|
+
|
|
152
|
+
def test_streaming_unsupported_provider_returns_501(self, client, monkeypatch):
|
|
153
|
+
"""Providers without chat_stream override must get a clear 501,
|
|
154
|
+
not a fake buffered stream."""
|
|
155
|
+
from tokenmizer.api import app as app_module
|
|
156
|
+
from tokenmizer.providers.providers import BaseProvider
|
|
157
|
+
|
|
158
|
+
class NoStreamProvider(BaseProvider):
|
|
159
|
+
async def _call(self, *a, **k): # pragma: no cover
|
|
160
|
+
raise NotImplementedError
|
|
161
|
+
|
|
162
|
+
monkeypatch.setattr(app_module, "_get_provider", lambda: NoStreamProvider())
|
|
130
163
|
c, _ = client
|
|
131
164
|
r = c.post("/v1/chat/completions", json={
|
|
132
165
|
"messages": [{"role": "user", "content": "hi"}],
|
|
@@ -16,7 +16,7 @@ from typing import Optional
|
|
|
16
16
|
|
|
17
17
|
from fastapi import Depends, FastAPI, HTTPException, Request
|
|
18
18
|
from fastapi.middleware.cors import CORSMiddleware
|
|
19
|
-
from fastapi.responses import HTMLResponse
|
|
19
|
+
from fastapi.responses import HTMLResponse, StreamingResponse
|
|
20
20
|
from pydantic import BaseModel
|
|
21
21
|
|
|
22
22
|
from tokenmizer.analytics.engine import AnalyticsEngine
|
|
@@ -261,7 +261,7 @@ async def lifespan(app: FastAPI):
|
|
|
261
261
|
app = FastAPI(
|
|
262
262
|
title="TokenMizer",
|
|
263
263
|
description="Never lose your AI context again.",
|
|
264
|
-
version="0.
|
|
264
|
+
version="0.3.0",
|
|
265
265
|
lifespan=lifespan,
|
|
266
266
|
docs_url="/docs",
|
|
267
267
|
redoc_url="/redoc",
|
|
@@ -558,17 +558,6 @@ async def _call_provider(
|
|
|
558
558
|
output_tokens = count_tokens(cached.response, model)
|
|
559
559
|
return cached.response, 0, output_tokens, 0.0
|
|
560
560
|
|
|
561
|
-
# Streaming check
|
|
562
|
-
if req.stream:
|
|
563
|
-
raise HTTPException(
|
|
564
|
-
status_code=501,
|
|
565
|
-
detail=(
|
|
566
|
-
"Streaming is not yet supported by the TokenMizer proxy. "
|
|
567
|
-
"Set stream=False in your request, or connect directly to "
|
|
568
|
-
"your provider for streaming. True SSE streaming is planned for v0.3."
|
|
569
|
-
),
|
|
570
|
-
)
|
|
571
|
-
|
|
572
561
|
# LLM call
|
|
573
562
|
# NOTE: `messages` is already redacted — redaction now happens once at
|
|
574
563
|
# ingestion in chat_completions() so every downstream consumer (this call,
|
|
@@ -609,6 +598,83 @@ async def _call_provider(
|
|
|
609
598
|
return response_text, input_tokens, output_tokens, latency_ms
|
|
610
599
|
|
|
611
600
|
|
|
601
|
+
def _stream_response(req, messages, model, user_content, session_id,
|
|
602
|
+
savings, orig_input_tokens) -> StreamingResponse:
|
|
603
|
+
"""True SSE passthrough (OpenAI chat.completion.chunk format).
|
|
604
|
+
|
|
605
|
+
Cache hits stream as a single chunk. After the stream closes, analytics
|
|
606
|
+
and the semantic-cache write run on the accumulated text — same
|
|
607
|
+
bookkeeping as the non-stream path, minus output trimming.
|
|
608
|
+
"""
|
|
609
|
+
import json as _json
|
|
610
|
+
|
|
611
|
+
from tokenmizer.providers.providers import BaseProvider, ProviderError
|
|
612
|
+
|
|
613
|
+
provider = _get_provider()
|
|
614
|
+
if getattr(type(provider), "chat_stream", None) is BaseProvider.chat_stream:
|
|
615
|
+
raise HTTPException(
|
|
616
|
+
status_code=501,
|
|
617
|
+
detail=(f"Streaming passthrough is not implemented for provider "
|
|
618
|
+
f"'{settings.provider}' yet (supported: anthropic, openai, "
|
|
619
|
+
f"deepseek, mistral, openrouter, grok, ollama). "
|
|
620
|
+
f"Set stream=false for this provider."),
|
|
621
|
+
)
|
|
622
|
+
|
|
623
|
+
resp_id = f"chatcmpl-{uuid.uuid4().hex[:12]}"
|
|
624
|
+
created = int(time.time())
|
|
625
|
+
|
|
626
|
+
def _chunk(delta: dict, finish: str | None = None) -> str:
|
|
627
|
+
return "data: " + _json.dumps({
|
|
628
|
+
"id": resp_id, "object": "chat.completion.chunk",
|
|
629
|
+
"created": created, "model": model, "session_id": session_id,
|
|
630
|
+
"choices": [{"index": 0, "delta": delta, "finish_reason": finish}],
|
|
631
|
+
}) + "\n\n"
|
|
632
|
+
|
|
633
|
+
async def _gen():
|
|
634
|
+
full_text = ""
|
|
635
|
+
t0 = time.monotonic()
|
|
636
|
+
yield _chunk({"role": "assistant"})
|
|
637
|
+
try:
|
|
638
|
+
cached = (_cache.get(user_content, session_id=session_id)
|
|
639
|
+
if settings.cache.enabled and user_content else None)
|
|
640
|
+
if cached:
|
|
641
|
+
full_text = cached.response
|
|
642
|
+
yield _chunk({"content": full_text})
|
|
643
|
+
else:
|
|
644
|
+
async for piece in provider.chat_stream(
|
|
645
|
+
messages=messages, model=model,
|
|
646
|
+
max_tokens=req.max_tokens or 4096,
|
|
647
|
+
**_sampling_kwargs(req),
|
|
648
|
+
):
|
|
649
|
+
full_text += piece
|
|
650
|
+
yield _chunk({"content": piece})
|
|
651
|
+
except ProviderError as e:
|
|
652
|
+
# Mid-stream failure: SSE can't change the status code anymore —
|
|
653
|
+
# emit an explicit error event instead of silently truncating.
|
|
654
|
+
yield "data: " + _json.dumps({"error": {"message": str(e), "type": "provider_error"}}) + "\n\n"
|
|
655
|
+
yield _chunk({}, finish="stop")
|
|
656
|
+
yield "data: [DONE]\n\n"
|
|
657
|
+
|
|
658
|
+
# Post-stream bookkeeping
|
|
659
|
+
latency_ms = (time.monotonic() - t0) * 1000
|
|
660
|
+
output_tokens = count_tokens(full_text, model)
|
|
661
|
+
input_tokens = count_messages_tokens(messages, model)
|
|
662
|
+
if settings.cache.enabled and user_content and full_text:
|
|
663
|
+
_cache.set(user_content, full_text, input_tokens=input_tokens,
|
|
664
|
+
output_tokens=output_tokens, session_id=session_id)
|
|
665
|
+
_analytics.record(
|
|
666
|
+
session_id=session_id, provider=settings.provider, model=model,
|
|
667
|
+
input_tokens_original=orig_input_tokens,
|
|
668
|
+
input_tokens_sent=input_tokens, output_tokens=output_tokens,
|
|
669
|
+
tokens_saved=sum(savings.values()), latency_ms=latency_ms,
|
|
670
|
+
cache_hit=False, layer_savings=savings,
|
|
671
|
+
)
|
|
672
|
+
|
|
673
|
+
return StreamingResponse(_gen(), media_type="text/event-stream",
|
|
674
|
+
headers={"Cache-Control": "no-cache",
|
|
675
|
+
"X-Accel-Buffering": "no"})
|
|
676
|
+
|
|
677
|
+
|
|
612
678
|
@app.post("/v1/chat/completions", dependencies=[Depends(verify_api_key), Depends(injection_guard)])
|
|
613
679
|
async def chat_completions(req: ChatRequest, request: Request):
|
|
614
680
|
"""
|
|
@@ -662,6 +728,13 @@ async def chat_completions(req: ChatRequest, request: Request):
|
|
|
662
728
|
session_id, graph, raw_messages, messages, model, savings, user_query
|
|
663
729
|
)
|
|
664
730
|
|
|
731
|
+
# Streaming: true SSE passthrough (v0.3). Output-trimming is skipped in
|
|
732
|
+
# stream mode (can't trim tokens that already left the building) — all
|
|
733
|
+
# input-side layers (file intel, compression, graph context) still apply.
|
|
734
|
+
if req.stream:
|
|
735
|
+
return _stream_response(req, messages, model, user_content,
|
|
736
|
+
session_id, savings, orig_input_tokens)
|
|
737
|
+
|
|
665
738
|
# Layer 5: call provider (or return cache hit)
|
|
666
739
|
response_text, input_tokens_actual, output_tokens, latency_ms = await _call_provider(
|
|
667
740
|
req, messages, model, user_content, session_id, savings
|
|
@@ -107,6 +107,21 @@ class BaseProvider(ABC):
|
|
|
107
107
|
|
|
108
108
|
raise ProviderError(self.__class__.__name__, "max_retries", "All retry attempts exhausted", retryable=False)
|
|
109
109
|
|
|
110
|
+
def chat_stream(self, messages: list[dict], model: str = "",
|
|
111
|
+
max_tokens: int = 4096, system: str = "", **kwargs):
|
|
112
|
+
"""Async generator yielding text chunks as the provider produces them.
|
|
113
|
+
|
|
114
|
+
Providers that support true streaming override this. The base
|
|
115
|
+
implementation raises so the API layer can return a clear 501 for
|
|
116
|
+
providers where passthrough streaming isn't implemented yet, instead
|
|
117
|
+
of silently degrading to a fake (buffered) stream.
|
|
118
|
+
"""
|
|
119
|
+
raise ProviderError(
|
|
120
|
+
self.__class__.__name__, "stream_not_supported",
|
|
121
|
+
f"Streaming passthrough not implemented for {self.__class__.__name__}",
|
|
122
|
+
retryable=False,
|
|
123
|
+
)
|
|
124
|
+
|
|
110
125
|
|
|
111
126
|
# ── Anthropic ─────────────────────────────────────────────────────────────────
|
|
112
127
|
|
|
@@ -179,6 +194,37 @@ class AnthropicProvider(BaseProvider):
|
|
|
179
194
|
retryable = e.status_code in (500, 502, 503, 529)
|
|
180
195
|
raise ProviderError("anthropic", f"http_{e.status_code}", str(e), retryable=retryable)
|
|
181
196
|
|
|
197
|
+
async def chat_stream(self, messages: list[dict], model: str = "",
|
|
198
|
+
max_tokens: int = 4096, system: str = "", **kwargs):
|
|
199
|
+
"""True SSE passthrough — yields text chunks as Anthropic streams them."""
|
|
200
|
+
try:
|
|
201
|
+
import anthropic
|
|
202
|
+
except ImportError:
|
|
203
|
+
raise ImportError("pip install anthropic")
|
|
204
|
+
model = model or self.default_model
|
|
205
|
+
client = anthropic.AsyncAnthropic(api_key=self.api_key)
|
|
206
|
+
|
|
207
|
+
sys_parts = [m["content"] for m in messages if m.get("role") == "system"]
|
|
208
|
+
if system:
|
|
209
|
+
sys_parts.insert(0, system)
|
|
210
|
+
conv = [m for m in messages if m.get("role") != "system"]
|
|
211
|
+
kwargs_clean = _sampling(kwargs)
|
|
212
|
+
if "stop" in kwargs_clean:
|
|
213
|
+
kwargs_clean["stop_sequences"] = _as_stop_list(kwargs_clean.pop("stop"))
|
|
214
|
+
if sys_parts:
|
|
215
|
+
kwargs_clean["system"] = "\n\n".join(sys_parts)
|
|
216
|
+
|
|
217
|
+
try:
|
|
218
|
+
async with client.messages.stream(
|
|
219
|
+
model=model, messages=conv, max_tokens=max_tokens, **kwargs_clean
|
|
220
|
+
) as s:
|
|
221
|
+
async for chunk in s.text_stream:
|
|
222
|
+
yield chunk
|
|
223
|
+
except anthropic.RateLimitError as e:
|
|
224
|
+
raise ProviderError("anthropic", "rate_limit", str(e), retryable=True)
|
|
225
|
+
except anthropic.APIStatusError as e:
|
|
226
|
+
raise ProviderError("anthropic", f"http_{e.status_code}", str(e), retryable=False)
|
|
227
|
+
|
|
182
228
|
|
|
183
229
|
# ── OpenAI ────────────────────────────────────────────────────────────────────
|
|
184
230
|
|
|
@@ -237,6 +283,37 @@ class OpenAIProvider(BaseProvider):
|
|
|
237
283
|
retryable = "rate" in err or "server" in err or "timeout" in err
|
|
238
284
|
raise ProviderError("openai", "api_error", str(e), retryable=retryable)
|
|
239
285
|
|
|
286
|
+
async def chat_stream(self, messages: list[dict], model: str = "",
|
|
287
|
+
max_tokens: int = 4096, system: str = "", **kwargs):
|
|
288
|
+
"""True SSE passthrough for OpenAI and all OpenAI-compatible providers
|
|
289
|
+
(DeepSeek, Mistral, OpenRouter, Grok inherit this)."""
|
|
290
|
+
try:
|
|
291
|
+
from openai import AsyncOpenAI
|
|
292
|
+
except ImportError:
|
|
293
|
+
raise ImportError("pip install openai")
|
|
294
|
+
model = model or self.default_model
|
|
295
|
+
client = AsyncOpenAI(
|
|
296
|
+
api_key=self.api_key,
|
|
297
|
+
**({"base_url": self._base_url} if self._base_url else {}),
|
|
298
|
+
)
|
|
299
|
+
all_messages = messages[:]
|
|
300
|
+
if system:
|
|
301
|
+
all_messages = [{"role": "system", "content": system}] + all_messages
|
|
302
|
+
try:
|
|
303
|
+
stream = await client.chat.completions.create(
|
|
304
|
+
model=model, messages=all_messages, max_tokens=max_tokens,
|
|
305
|
+
stream=True, **_sampling(kwargs),
|
|
306
|
+
)
|
|
307
|
+
async for chunk in stream:
|
|
308
|
+
delta = chunk.choices[0].delta.content if chunk.choices else None
|
|
309
|
+
if delta:
|
|
310
|
+
yield delta
|
|
311
|
+
except Exception as e:
|
|
312
|
+
err = str(e).lower()
|
|
313
|
+
retryable = "rate" in err or "server" in err or "timeout" in err
|
|
314
|
+
raise ProviderError(self.__class__.__name__.lower().replace("provider", ""),
|
|
315
|
+
"api_error", str(e), retryable=retryable)
|
|
316
|
+
|
|
240
317
|
|
|
241
318
|
# ── DeepSeek ──────────────────────────────────────────────────────────────────
|
|
242
319
|
|
|
@@ -427,6 +504,37 @@ class OllamaProvider(BaseProvider):
|
|
|
427
504
|
except Exception as e:
|
|
428
505
|
raise ProviderError("ollama", "api_error", str(e), retryable=True)
|
|
429
506
|
|
|
507
|
+
async def chat_stream(self, messages: list[dict], model: str = "",
|
|
508
|
+
max_tokens: int = 4096, system: str = "", **kwargs):
|
|
509
|
+
"""Streaming passthrough for local Ollama models."""
|
|
510
|
+
import json as _json
|
|
511
|
+
|
|
512
|
+
import httpx
|
|
513
|
+
model = model or self.default_model
|
|
514
|
+
all_messages = messages[:]
|
|
515
|
+
if system:
|
|
516
|
+
all_messages = [{"role": "system", "content": system}] + all_messages
|
|
517
|
+
payload = {"model": model, "messages": all_messages, "stream": True,
|
|
518
|
+
"options": {"num_predict": max_tokens}}
|
|
519
|
+
try:
|
|
520
|
+
async with httpx.AsyncClient(timeout=120) as client:
|
|
521
|
+
async with client.stream("POST", f"{self._base_url}/api/chat",
|
|
522
|
+
json=payload) as r:
|
|
523
|
+
r.raise_for_status()
|
|
524
|
+
async for line in r.aiter_lines():
|
|
525
|
+
if not line.strip():
|
|
526
|
+
continue
|
|
527
|
+
data = _json.loads(line)
|
|
528
|
+
chunk = data.get("message", {}).get("content", "")
|
|
529
|
+
if chunk:
|
|
530
|
+
yield chunk
|
|
531
|
+
if data.get("done"):
|
|
532
|
+
break
|
|
533
|
+
except ProviderError:
|
|
534
|
+
raise
|
|
535
|
+
except Exception as e:
|
|
536
|
+
raise ProviderError("ollama", "api_error", str(e), retryable=True)
|
|
537
|
+
|
|
430
538
|
|
|
431
539
|
# ── Registry ──────────────────────────────────────────────────────────────────
|
|
432
540
|
|
tokenmizer-0.2.6/server.json
DELETED
|
@@ -1,29 +0,0 @@
|
|
|
1
|
-
{
|
|
2
|
-
"$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
|
|
3
|
-
"name": "io.github.Shweta-Mishra-ai/tokenmizer",
|
|
4
|
-
"description": "An MCP server that provides [describe what your server does]",
|
|
5
|
-
"repository": {
|
|
6
|
-
"url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
|
|
7
|
-
"source": "github"
|
|
8
|
-
},
|
|
9
|
-
"version": "1.0.0",
|
|
10
|
-
"packages": [
|
|
11
|
-
{
|
|
12
|
-
"registryType": "pypi",
|
|
13
|
-
"identifier": "tokenmizer\"\r",
|
|
14
|
-
"version": "1.0.0",
|
|
15
|
-
"transport": {
|
|
16
|
-
"type": "stdio"
|
|
17
|
-
},
|
|
18
|
-
"environmentVariables": [
|
|
19
|
-
{
|
|
20
|
-
"description": "Your API key for the service",
|
|
21
|
-
"isRequired": true,
|
|
22
|
-
"format": "string",
|
|
23
|
-
"isSecret": true,
|
|
24
|
-
"name": "YOUR_API_KEY"
|
|
25
|
-
}
|
|
26
|
-
]
|
|
27
|
-
}
|
|
28
|
-
]
|
|
29
|
-
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|