tokenmizer 0.2.6__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/CHANGELOG.md +15 -0
  2. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/PKG-INFO +57 -14
  3. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/README.md +56 -13
  4. tokenmizer-0.3.0/docs/assets/demo.gif +0 -0
  5. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/pyproject.toml +1 -1
  6. tokenmizer-0.3.0/scripts/gen_demo_gif.py +108 -0
  7. tokenmizer-0.3.0/server.json +34 -0
  8. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/test_api_endpoint.py +34 -1
  9. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/__init__.py +1 -1
  10. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/app.py +86 -13
  11. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/providers/providers.py +108 -0
  12. tokenmizer-0.2.6/server.json +0 -29
  13. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/marketplace.json +0 -0
  14. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/plugin.json +0 -0
  15. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/analyze/SKILL.md +0 -0
  16. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/checkpoint/SKILL.md +0 -0
  17. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/resume/SKILL.md +0 -0
  18. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.claude-plugin/skills/stats/SKILL.md +0 -0
  19. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
  20. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/ISSUE_TEMPLATE/extraction_miss.md +0 -0
  21. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  22. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/workflows/ci.yml +0 -0
  23. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.github/workflows/release.yml +0 -0
  24. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.gitignore +0 -0
  25. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/.mcp.json +0 -0
  26. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/CONTRIBUTING.md +0 -0
  27. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/Dockerfile +0 -0
  28. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/LICENSE +0 -0
  29. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/SECURITY.md +0 -0
  30. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/TESTING.md +0 -0
  31. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/USAGE.md +0 -0
  32. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/__init__.py +0 -0
  33. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/__init__.py +0 -0
  34. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner.py +0 -0
  35. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner_v2.py +0 -0
  36. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/checkpoint_accuracy/runner_v3.py +0 -0
  37. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/graph_retrieval/__init__.py +0 -0
  38. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/graph_retrieval/runner.py +0 -0
  39. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/latency/__init__.py +0 -0
  40. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/latency/runner.py +0 -0
  41. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/benchmarks/resume_quality/__init__.py +0 -0
  42. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docker-compose.yml +0 -0
  43. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docs/assets/architecture.svg +0 -0
  44. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/docs/assets/logo.svg +0 -0
  45. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/examples/basic_usage.py +0 -0
  46. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/install.sh +0 -0
  47. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/mcp_e2e_check.py +0 -0
  48. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/run_stdlib_tests.py +0 -0
  49. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/setup.sh +0 -0
  50. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/scripts/static_audit.py +0 -0
  51. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/__init__.py +0 -0
  52. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/chaos/__init__.py +0 -0
  53. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/chaos/test_recovery.py +0 -0
  54. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/conftest.py +0 -0
  55. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/__init__.py +0 -0
  56. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/integration/test_checkpoint.py +0 -0
  57. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/memory_accuracy/__init__.py +0 -0
  58. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/memory_accuracy/test_retention.py +0 -0
  59. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/__init__.py +0 -0
  60. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_cache.py +0 -0
  61. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_compression_correctness.py +0 -0
  62. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_decision_cache_async.py +0 -0
  63. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_file_intelligence.py +0 -0
  64. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_graph.py +0 -0
  65. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_graph_persistence.py +0 -0
  66. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_hybrid_extractor.py +0 -0
  67. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_rate_limiter.py +0 -0
  68. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_security.py +0 -0
  69. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_tokenizer.py +0 -0
  70. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tests/unit/test_validator.py +0 -0
  71. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/agents/__init__.py +0 -0
  72. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/analytics/__init__.py +0 -0
  73. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/analytics/engine.py +0 -0
  74. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/__init__.py +0 -0
  75. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/api/rate_limiter.py +0 -0
  76. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/checkpoints/__init__.py +0 -0
  77. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/checkpoints/manager.py +0 -0
  78. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/cli.py +0 -0
  79. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/__init__.py +0 -0
  80. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/engine.py +0 -0
  81. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/output_trimmer.py +0 -0
  82. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/compression/window.py +0 -0
  83. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/config/__init__.py +0 -0
  84. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/config/settings.py +0 -0
  85. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/__init__.py +0 -0
  86. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/dto.py +0 -0
  87. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/errors.py +0 -0
  88. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/core/tokenizer.py +0 -0
  89. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/dashboard/__init__.py +0 -0
  90. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/dashboard/page.py +0 -0
  91. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/filters/__init__.py +0 -0
  92. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/filters/file_intelligence.py +0 -0
  93. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/__init__.py +0 -0
  94. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/decision_tracker.py +0 -0
  95. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/graph.py +0 -0
  96. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/helpers.py +0 -0
  97. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/hybrid_extractor.py +0 -0
  98. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/types.py +0 -0
  99. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/validator.py +0 -0
  100. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/graph_memory/visualization.py +0 -0
  101. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/mcp/__init__.py +0 -0
  102. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/mcp/server.py +0 -0
  103. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/providers/__init__.py +0 -0
  104. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/__init__.py +0 -0
  105. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/auth.py +0 -0
  106. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/middleware.py +0 -0
  107. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/security/redaction.py +0 -0
  108. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/semantic_cache/__init__.py +0 -0
  109. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/semantic_cache/cache.py +0 -0
  110. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/state/__init__.py +0 -0
  111. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/state/backend.py +0 -0
  112. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer/storage/__init__.py +0 -0
  113. {tokenmizer-0.2.6 → tokenmizer-0.3.0}/tokenmizer.yaml +0 -0
@@ -1,5 +1,20 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.3.0] — 2026-07-02 — true SSE streaming
4
+
5
+ - **`stream: true` now works** — real passthrough streaming in OpenAI
6
+ `chat.completion.chunk` SSE format for Anthropic, OpenAI, DeepSeek,
7
+ Mistral, OpenRouter, Grok (OpenAI-compatible base) and Ollama. Cursor,
8
+ Continue.dev and every streaming client can now point at TokenMizer
9
+ without config changes.
10
+ - All input-side layers (file intelligence, compression, graph memory,
11
+ context injection) apply to streamed requests; output trimming is
12
+ skipped in stream mode by design. Cache hits stream as a single chunk;
13
+ post-stream analytics + cache writes preserved.
14
+ - Providers without passthrough support return an explicit 501 (never a
15
+ fake buffered stream). Mid-stream provider failures emit an SSE `error`
16
+ event instead of silently truncating.
17
+
3
18
  ## [0.2.6] — 2026-07-02 — MCP registry readiness
4
19
 
5
20
  - NEW console script `tokenmizer-mcp` — runs the MCP stdio server directly
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tokenmizer
3
- Version: 0.2.6
3
+ Version: 0.3.0
4
4
  Summary: Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session.
5
5
  Project-URL: Homepage, https://github.com/Shweta-Mishra-ai/tokenmizer
6
6
  Project-URL: Repository, https://github.com/Shweta-Mishra-ai/tokenmizer
@@ -87,12 +87,24 @@ Description-Content-Type: text/markdown
87
87
 
88
88
  <p>
89
89
  <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/v/tokenmizer?color=7c6af7&style=flat-square" alt="PyPI"/></a>
90
- <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/pyversions/tokenmizer?color=5ee7c8&style=flat-square"/></a>
90
+ <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/dm/tokenmizer?color=5ee7c8&style=flat-square" alt="Downloads"/></a>
91
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80" alt="CI"/></a>
92
+ <a href="https://registry.modelcontextprotocol.io/v0/servers?search=tokenmizer"><img src="https://img.shields.io/badge/MCP%20Registry-published-5ee7c8?style=flat-square" alt="MCP Registry"/></a>
91
93
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-4ade80?style=flat-square"/></a>
92
- <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80"/></a>
93
- <img src="https://img.shields.io/badge/Claude%20Code-Plugin-7c6af7?style=flat-square&logo=anthropic"/>
94
- <img src="https://img.shields.io/badge/MCP-Compatible-5ee7c8?style=flat-square"/>
94
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/stargazers"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=flat-square&color=f9d84a" alt="Stars"/></a>
95
95
  </p>
96
+
97
+ <p>
98
+ <a href="#quick-start"><b>Quick Start</b></a> ·
99
+ <a href="#how-tokenmizer-solves-it"><b>How it works</b></a> ·
100
+ <a href="#benchmarks"><b>Benchmarks</b></a> ·
101
+ <a href="#claude-code-integration"><b>Claude Code</b></a> ·
102
+ <a href="#contributing"><b>Contributing</b></a>
103
+ </p>
104
+
105
+ <img src="docs/assets/demo.gif" width="860" alt="TokenMizer demo: 40-turn session checkpointed at 87% context, resumed next day in 233 tokens"/>
106
+ <br/>
107
+ <sub>Real run: 25-node graph, checkpoint <code>ckpt_21a0959c3ddf</code>, 233-token resume. Regenerate with <code>python scripts/gen_demo_gif.py</code>.</sub>
96
108
  </div>
97
109
 
98
110
  ---
@@ -193,13 +205,9 @@ response = client.chat.completions.create(
193
205
  )
194
206
  ```
195
207
 
196
- > **⚠️ Streaming is not supported yet.** Set `stream: false` (or leave it unset —
197
- > that's the default) in every request. Requests with `stream: true` return
198
- > `HTTP 501`. True SSE streaming is planned for v0.3.
199
- >
200
- > **Cursor:** Settings → Models → your model entry → disable "Stream responses"
201
- > **Continue.dev:** in `config.json`, set `"streamResponses": false` for the
202
- > TokenMizer provider entry
208
+ > ✅ **Streaming works** (v0.3+): `stream: true` gives real SSE passthrough for
209
+ > Anthropic, OpenAI, DeepSeek, Mistral, OpenRouter, Grok and Ollama. Cursor and
210
+ > Continue.dev work with default settings — no config changes needed.
203
211
 
204
212
  ---
205
213
 
@@ -505,16 +513,49 @@ tokenmizer stats
505
513
 
506
514
  ---
507
515
 
516
+ ## Roadmap
517
+
518
+ | Version | Focus |
519
+ |---|---|
520
+ | **v0.3** | SSE streaming passthrough (checkpoint on stream close) |
521
+ | v0.4 | Cross-session memory · embedding-based edge linking |
522
+ | v0.5 | Per-node storage schema (scale past 200-node graphs) |
523
+ | Research | Real-transcript benchmark suite → paper ([tokenmizer-research](https://github.com/Shweta-Mishra-ai/tokenmizer-research)) |
524
+
525
+ Have a use case that doesn't fit? [Open an issue](https://github.com/Shweta-Mishra-ai/tokenmizer/issues/new/choose) — extraction misses have their own issue template.
526
+
527
+ ---
528
+
508
529
  ## Contributing
509
530
 
510
- See [CONTRIBUTING.md](CONTRIBUTING.md). Graph extraction contributions are the highest priority.
531
+ Contributions welcome — this project merges fast (median PR review < 1 day).
511
532
 
512
533
  ```bash
513
534
  git clone https://github.com/Shweta-Mishra-ai/tokenmizer
535
+ cd tokenmizer
514
536
  pip install -e ".[dev]"
515
- pytest tests/ -v && ruff check tokenmizer/
537
+ pytest tests/ -v && ruff check tokenmizer/ # 218 tests, must stay green
538
+ python scripts/mcp_e2e_check.py # full-pipeline e2e check
516
539
  ```
517
540
 
541
+ **Highest-impact areas right now:**
542
+
543
+ 1. **Graph extraction quality** — real-world transcripts where extraction misses tasks/decisions (file an [extraction-miss issue](.github/ISSUE_TEMPLATE/extraction_miss.md) even if you don't fix it — the failing transcript itself is the contribution)
544
+ 2. **SSE streaming** (v0.3 headline feature)
545
+ 3. **Benchmark sessions** — add a real session + ground truth to `benchmarks/`
546
+
547
+ Every PR runs the full CI gauntlet (tests × 3 Python versions, lint, Docker build). See [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines and [TESTING.md](TESTING.md) for the test architecture.
548
+
549
+ ---
550
+
551
+ ## Support the project
552
+
553
+ TokenMizer is built and maintained by one person. If it saved you tokens, time, or a lost session:
554
+
555
+ - ⭐ **[Star the repo](https://github.com/Shweta-Mishra-ai/tokenmizer)** — the single best way to help others find it
556
+ - 🐛 [Report a bug](https://github.com/Shweta-Mishra-ai/tokenmizer/issues) — especially extraction misses
557
+ - 📣 Share your before/after token numbers (`tokenmizer stats`) — real usage data shapes the roadmap
558
+
518
559
  ---
519
560
 
520
561
  ## License
@@ -525,4 +566,6 @@ MIT © [Shweta Mishra](https://github.com/Shweta-Mishra-ai)
525
566
 
526
567
  <div align="center">
527
568
  <sub>Built for developers who spend too much time re-explaining their projects to AI.</sub>
569
+ <br/><br/>
570
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=social" alt="GitHub stars"/></a>
528
571
  </div>
@@ -12,12 +12,24 @@
12
12
 
13
13
  <p>
14
14
  <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/v/tokenmizer?color=7c6af7&style=flat-square" alt="PyPI"/></a>
15
- <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/pyversions/tokenmizer?color=5ee7c8&style=flat-square"/></a>
15
+ <a href="https://pypi.org/project/tokenmizer"><img src="https://img.shields.io/pypi/dm/tokenmizer?color=5ee7c8&style=flat-square" alt="Downloads"/></a>
16
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80" alt="CI"/></a>
17
+ <a href="https://registry.modelcontextprotocol.io/v0/servers?search=tokenmizer"><img src="https://img.shields.io/badge/MCP%20Registry-published-5ee7c8?style=flat-square" alt="MCP Registry"/></a>
16
18
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-4ade80?style=flat-square"/></a>
17
- <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/actions"><img src="https://img.shields.io/github/actions/workflow/status/Shweta-Mishra-ai/tokenmizer/ci.yml?branch=main&style=flat-square&color=4ade80"/></a>
18
- <img src="https://img.shields.io/badge/Claude%20Code-Plugin-7c6af7?style=flat-square&logo=anthropic"/>
19
- <img src="https://img.shields.io/badge/MCP-Compatible-5ee7c8?style=flat-square"/>
19
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer/stargazers"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=flat-square&color=f9d84a" alt="Stars"/></a>
20
20
  </p>
21
+
22
+ <p>
23
+ <a href="#quick-start"><b>Quick Start</b></a> ·
24
+ <a href="#how-tokenmizer-solves-it"><b>How it works</b></a> ·
25
+ <a href="#benchmarks"><b>Benchmarks</b></a> ·
26
+ <a href="#claude-code-integration"><b>Claude Code</b></a> ·
27
+ <a href="#contributing"><b>Contributing</b></a>
28
+ </p>
29
+
30
+ <img src="docs/assets/demo.gif" width="860" alt="TokenMizer demo: 40-turn session checkpointed at 87% context, resumed next day in 233 tokens"/>
31
+ <br/>
32
+ <sub>Real run: 25-node graph, checkpoint <code>ckpt_21a0959c3ddf</code>, 233-token resume. Regenerate with <code>python scripts/gen_demo_gif.py</code>.</sub>
21
33
  </div>
22
34
 
23
35
  ---
@@ -118,13 +130,9 @@ response = client.chat.completions.create(
118
130
  )
119
131
  ```
120
132
 
121
- > **⚠️ Streaming is not supported yet.** Set `stream: false` (or leave it unset —
122
- > that's the default) in every request. Requests with `stream: true` return
123
- > `HTTP 501`. True SSE streaming is planned for v0.3.
124
- >
125
- > **Cursor:** Settings → Models → your model entry → disable "Stream responses"
126
- > **Continue.dev:** in `config.json`, set `"streamResponses": false` for the
127
- > TokenMizer provider entry
133
+ > ✅ **Streaming works** (v0.3+): `stream: true` gives real SSE passthrough for
134
+ > Anthropic, OpenAI, DeepSeek, Mistral, OpenRouter, Grok and Ollama. Cursor and
135
+ > Continue.dev work with default settings — no config changes needed.
128
136
 
129
137
  ---
130
138
 
@@ -430,16 +438,49 @@ tokenmizer stats
430
438
 
431
439
  ---
432
440
 
441
+ ## Roadmap
442
+
443
+ | Version | Focus |
444
+ |---|---|
445
+ | **v0.3** | SSE streaming passthrough (checkpoint on stream close) |
446
+ | v0.4 | Cross-session memory · embedding-based edge linking |
447
+ | v0.5 | Per-node storage schema (scale past 200-node graphs) |
448
+ | Research | Real-transcript benchmark suite → paper ([tokenmizer-research](https://github.com/Shweta-Mishra-ai/tokenmizer-research)) |
449
+
450
+ Have a use case that doesn't fit? [Open an issue](https://github.com/Shweta-Mishra-ai/tokenmizer/issues/new/choose) — extraction misses have their own issue template.
451
+
452
+ ---
453
+
433
454
  ## Contributing
434
455
 
435
- See [CONTRIBUTING.md](CONTRIBUTING.md). Graph extraction contributions are the highest priority.
456
+ Contributions welcome — this project merges fast (median PR review < 1 day).
436
457
 
437
458
  ```bash
438
459
  git clone https://github.com/Shweta-Mishra-ai/tokenmizer
460
+ cd tokenmizer
439
461
  pip install -e ".[dev]"
440
- pytest tests/ -v && ruff check tokenmizer/
462
+ pytest tests/ -v && ruff check tokenmizer/ # 218 tests, must stay green
463
+ python scripts/mcp_e2e_check.py # full-pipeline e2e check
441
464
  ```
442
465
 
466
+ **Highest-impact areas right now:**
467
+
468
+ 1. **Graph extraction quality** — real-world transcripts where extraction misses tasks/decisions (file an [extraction-miss issue](.github/ISSUE_TEMPLATE/extraction_miss.md) even if you don't fix it — the failing transcript itself is the contribution)
469
+ 2. **SSE streaming** (v0.3 headline feature)
470
+ 3. **Benchmark sessions** — add a real session + ground truth to `benchmarks/`
471
+
472
+ Every PR runs the full CI gauntlet (tests × 3 Python versions, lint, Docker build). See [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines and [TESTING.md](TESTING.md) for the test architecture.
473
+
474
+ ---
475
+
476
+ ## Support the project
477
+
478
+ TokenMizer is built and maintained by one person. If it saved you tokens, time, or a lost session:
479
+
480
+ - ⭐ **[Star the repo](https://github.com/Shweta-Mishra-ai/tokenmizer)** — the single best way to help others find it
481
+ - 🐛 [Report a bug](https://github.com/Shweta-Mishra-ai/tokenmizer/issues) — especially extraction misses
482
+ - 📣 Share your before/after token numbers (`tokenmizer stats`) — real usage data shapes the roadmap
483
+
443
484
  ---
444
485
 
445
486
  ## License
@@ -450,4 +491,6 @@ MIT © [Shweta Mishra](https://github.com/Shweta-Mishra-ai)
450
491
 
451
492
  <div align="center">
452
493
  <sub>Built for developers who spend too much time re-explaining their projects to AI.</sub>
494
+ <br/><br/>
495
+ <a href="https://github.com/Shweta-Mishra-ai/tokenmizer"><img src="https://img.shields.io/github/stars/Shweta-Mishra-ai/tokenmizer?style=social" alt="GitHub stars"/></a>
453
496
  </div>
Binary file
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "tokenmizer"
7
- version = "0.2.6"
7
+ version = "0.3.0"
8
8
  description = "Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -0,0 +1,108 @@
1
+ # -*- coding: utf-8 -*-
2
+ """Generate docs/assets/demo.gif — terminal-style demo.
3
+
4
+ All numbers/outputs shown are REAL, captured from an actual run of
5
+ GraphMemory + CheckpointManager on the 40-turn-scale demo session
6
+ (checkpoint ckpt_21a0959c3ddf, 25 nodes, 233-token resume).
7
+ Run: python scripts/gen_demo_gif.py [output_path]
8
+ """
9
+ import os
10
+ import sys
11
+
12
+ from PIL import Image, ImageDraw, ImageFont
13
+
14
+ OUT = sys.argv[1] if len(sys.argv) > 1 else "docs/assets/demo.gif"
15
+ W, H = 920, 500
16
+ BG, FG = (13, 17, 23), (201, 209, 217)
17
+ GREEN, PURPLE, YELLOW = (63, 185, 80), (137, 87, 229), (210, 153, 34)
18
+ DIM, CYAN = (110, 118, 129), (57, 197, 187)
19
+
20
+ font = ImageFont.truetype("consola.ttf", 17)
21
+ bold = ImageFont.truetype("consolab.ttf", 17)
22
+ big = ImageFont.truetype("consolab.ttf", 26)
23
+
24
+
25
+ def frame(lines, title="tokenmizer - demo"):
26
+ im = Image.new("RGB", (W, H), BG)
27
+ d = ImageDraw.Draw(im)
28
+ d.rounded_rectangle([8, 8, W - 8, H - 8], 10, outline=(48, 54, 61), width=2)
29
+ for i, c in enumerate([(255, 95, 86), (255, 189, 46), (39, 201, 63)]):
30
+ d.ellipse([24 + i * 24, 20, 38 + i * 24, 34], fill=c)
31
+ d.text((W // 2 - 100, 18), title, font=font, fill=DIM)
32
+ y = 56
33
+ for txt, col, f in lines:
34
+ d.text((28, y), txt, font=f, fill=col)
35
+ y += 24
36
+ return im
37
+
38
+
39
+ frames, durs = [], []
40
+
41
+
42
+ def add(lines, ms):
43
+ frames.append(frame(lines))
44
+ durs.append(ms)
45
+
46
+
47
+ add([("$ pip install tokenmizer", FG, bold),
48
+ ("Successfully installed tokenmizer-0.2.6", GREEN, font),
49
+ ("", FG, font),
50
+ ("$ export TOKENMIZER_ANTHROPIC_API_KEY=sk-ant-...", FG, bold),
51
+ ("$ tokenmizer serve", FG, bold),
52
+ (" Proxy: http://localhost:8000/v1/chat/completions", CYAN, font),
53
+ (" Dashboard: http://localhost:8000", CYAN, font),
54
+ ("", FG, font),
55
+ ("# point any OpenAI-compatible client at localhost:8000 -", DIM, font),
56
+ ("# one line changed, zero code rewritten", DIM, font)], 2600)
57
+
58
+ add([("session: fastapi-auth model: claude-fable-5", DIM, font),
59
+ ("", FG, font),
60
+ ("you > Let's build a FastAPI auth service with JWT + PostgreSQL", FG, font),
61
+ ("ai > Decided: PostgreSQL (concurrent writes). Files: api/auth.py ...", DIM, font),
62
+ ("you > Use bcrypt and Python 3.12", FG, font),
63
+ ("ai > Decided: bcrypt (industry standard) ...", DIM, font),
64
+ ("you > Login returns 422, fix it", FG, font),
65
+ ("ai > Fixed: missing email validation in LoginRequest ...", DIM, font),
66
+ ("you > Add refresh tokens with Redis", FG, font),
67
+ ("ai > Decided: Redis for refresh tokens (faster revocation) ...", DIM, font),
68
+ (" ... 40 turns later ...", YELLOW, bold),
69
+ ("", FG, font),
70
+ (" context window: 87% full", YELLOW, bold)], 3200)
71
+
72
+ add([(" context window: 87% full", YELLOW, bold),
73
+ ("", FG, font),
74
+ (" * auto-checkpoint triggered *", PURPLE, bold),
75
+ ("", FG, font),
76
+ (" graph: 25 nodes extracted (tasks / decisions / files / errors)", FG, font),
77
+ (" checkpoint: ckpt_21a0959c3ddf saved to SQLite", GREEN, font),
78
+ (" resume size: 233 tokens", GREEN, bold),
79
+ ("", FG, font),
80
+ (" session history: ~5,800 tokens -> 233 tokens", CYAN, bold)], 3000)
81
+
82
+ add([(" -- next day, fresh session --", DIM, font),
83
+ ("", FG, font),
84
+ ("$ tokenmizer resume fastapi-auth", FG, bold),
85
+ ("", FG, font),
86
+ ("Goal: a FastAPI auth service with JWT and PostgreSQL", GREEN, font),
87
+ ("Working on: rate limiting with slowapi", FG, font),
88
+ ("Done: tests/test_auth.py (12 tests passing) | 422 login fix", FG, font),
89
+ ("Decided: PostgreSQL | bcrypt | Python 3.12 |", FG, font),
90
+ (" Redis for refresh tokens (faster revocation)", FG, font),
91
+ ("Changes: 'Use Redis' -> 'Redis for refresh token storage'", YELLOW, font),
92
+ ("Continue from: Write the tests", CYAN, font),
93
+ ("", FG, font),
94
+ (" [233 tokens - paste into any new session and keep going]", DIM, font)], 4200)
95
+
96
+ im = Image.new("RGB", (W, H), BG)
97
+ d = ImageDraw.Draw(im)
98
+ d.rounded_rectangle([8, 8, W - 8, H - 8], 10, outline=(48, 54, 61), width=2)
99
+ d.text((W // 2 - 260, H // 2 - 70), "never re-explain your project again", font=big, fill=FG)
100
+ d.text((W // 2 - 215, H // 2 - 20), "~5,800 tokens -> 233 tokens resume", font=big, fill=GREEN)
101
+ d.text((W // 2 - 130, H // 2 + 40), "pip install tokenmizer", font=bold, fill=PURPLE)
102
+ frames.append(im)
103
+ durs.append(3500)
104
+
105
+ os.makedirs(os.path.dirname(OUT), exist_ok=True)
106
+ frames[0].save(OUT, save_all=True, append_images=frames[1:],
107
+ duration=durs, loop=0, optimize=True)
108
+ print("demo.gif:", os.path.getsize(OUT) // 1024, "KB,", len(frames), "frames")
@@ -0,0 +1,34 @@
1
+ {
2
+ "$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
3
+ "name": "io.github.Shweta-Mishra-ai/tokenmizer",
4
+ "description": "Graph-backed session memory for LLMs: checkpoint, resume, file analysis, token-savings stats.",
5
+ "repository": {
6
+ "url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
7
+ "source": "github"
8
+ },
9
+ "version": "0.2.6",
10
+ "packages": [
11
+ {
12
+ "registryType": "pypi",
13
+ "identifier": "tokenmizer",
14
+ "version": "0.2.6",
15
+ "transport": { "type": "stdio" },
16
+ "environmentVariables": [
17
+ {
18
+ "name": "TOKENMIZER_URL",
19
+ "description": "Base URL of the running TokenMizer proxy (default http://localhost:8000)",
20
+ "isRequired": false,
21
+ "format": "string",
22
+ "isSecret": false
23
+ },
24
+ {
25
+ "name": "TOKENMIZER_API_KEY",
26
+ "description": "API key if the proxy has auth enabled",
27
+ "isRequired": false,
28
+ "format": "string",
29
+ "isSecret": true
30
+ }
31
+ ]
32
+ }
33
+ ]
34
+ }
@@ -126,7 +126,40 @@ class TestChatCompletionsE2E:
126
126
  assert kwargs.get("temperature") == 0.0
127
127
  assert kwargs.get("top_p") == 0.9
128
128
 
129
- def test_stream_returns_501_with_clear_message(self, client):
129
+ def test_streaming_returns_sse_chunks(self, client):
130
+ """v0.3: stream=true must return true SSE passthrough in OpenAI
131
+ chat.completion.chunk format, ending with data: [DONE]."""
132
+ c, fake = client
133
+
134
+ async def fake_stream(**kwargs):
135
+ for piece in ("Hello", " streamed", " world"):
136
+ yield piece
137
+
138
+ fake.chat_stream = fake_stream
139
+ r = c.post("/v1/chat/completions", json={
140
+ "messages": [{"role": "user", "content": "stream me a story please"}],
141
+ "stream": True,
142
+ })
143
+ assert r.status_code == 200, r.text
144
+ assert "text/event-stream" in r.headers["content-type"]
145
+ body = r.text
146
+ assert "chat.completion.chunk" in body
147
+ assert '"content": "Hello"' in body
148
+ assert '"content": " streamed"' in body
149
+ assert '"finish_reason": "stop"' in body
150
+ assert body.rstrip().endswith("data: [DONE]")
151
+
152
+ def test_streaming_unsupported_provider_returns_501(self, client, monkeypatch):
153
+ """Providers without chat_stream override must get a clear 501,
154
+ not a fake buffered stream."""
155
+ from tokenmizer.api import app as app_module
156
+ from tokenmizer.providers.providers import BaseProvider
157
+
158
+ class NoStreamProvider(BaseProvider):
159
+ async def _call(self, *a, **k): # pragma: no cover
160
+ raise NotImplementedError
161
+
162
+ monkeypatch.setattr(app_module, "_get_provider", lambda: NoStreamProvider())
130
163
  c, _ = client
131
164
  r = c.post("/v1/chat/completions", json={
132
165
  "messages": [{"role": "user", "content": "hi"}],
@@ -1,6 +1,6 @@
1
1
  """TokenMizer — Never lose your AI context again."""
2
2
 
3
- __version__ = "0.2.6"
3
+ __version__ = "0.3.0"
4
4
  __all__ = ["GraphMemory", "CheckpointManager", "get_settings"]
5
5
 
6
6
 
@@ -16,7 +16,7 @@ from typing import Optional
16
16
 
17
17
  from fastapi import Depends, FastAPI, HTTPException, Request
18
18
  from fastapi.middleware.cors import CORSMiddleware
19
- from fastapi.responses import HTMLResponse
19
+ from fastapi.responses import HTMLResponse, StreamingResponse
20
20
  from pydantic import BaseModel
21
21
 
22
22
  from tokenmizer.analytics.engine import AnalyticsEngine
@@ -261,7 +261,7 @@ async def lifespan(app: FastAPI):
261
261
  app = FastAPI(
262
262
  title="TokenMizer",
263
263
  description="Never lose your AI context again.",
264
- version="0.2.6",
264
+ version="0.3.0",
265
265
  lifespan=lifespan,
266
266
  docs_url="/docs",
267
267
  redoc_url="/redoc",
@@ -558,17 +558,6 @@ async def _call_provider(
558
558
  output_tokens = count_tokens(cached.response, model)
559
559
  return cached.response, 0, output_tokens, 0.0
560
560
 
561
- # Streaming check
562
- if req.stream:
563
- raise HTTPException(
564
- status_code=501,
565
- detail=(
566
- "Streaming is not yet supported by the TokenMizer proxy. "
567
- "Set stream=False in your request, or connect directly to "
568
- "your provider for streaming. True SSE streaming is planned for v0.3."
569
- ),
570
- )
571
-
572
561
  # LLM call
573
562
  # NOTE: `messages` is already redacted — redaction now happens once at
574
563
  # ingestion in chat_completions() so every downstream consumer (this call,
@@ -609,6 +598,83 @@ async def _call_provider(
609
598
  return response_text, input_tokens, output_tokens, latency_ms
610
599
 
611
600
 
601
+ def _stream_response(req, messages, model, user_content, session_id,
602
+ savings, orig_input_tokens) -> StreamingResponse:
603
+ """True SSE passthrough (OpenAI chat.completion.chunk format).
604
+
605
+ Cache hits stream as a single chunk. After the stream closes, analytics
606
+ and the semantic-cache write run on the accumulated text — same
607
+ bookkeeping as the non-stream path, minus output trimming.
608
+ """
609
+ import json as _json
610
+
611
+ from tokenmizer.providers.providers import BaseProvider, ProviderError
612
+
613
+ provider = _get_provider()
614
+ if getattr(type(provider), "chat_stream", None) is BaseProvider.chat_stream:
615
+ raise HTTPException(
616
+ status_code=501,
617
+ detail=(f"Streaming passthrough is not implemented for provider "
618
+ f"'{settings.provider}' yet (supported: anthropic, openai, "
619
+ f"deepseek, mistral, openrouter, grok, ollama). "
620
+ f"Set stream=false for this provider."),
621
+ )
622
+
623
+ resp_id = f"chatcmpl-{uuid.uuid4().hex[:12]}"
624
+ created = int(time.time())
625
+
626
+ def _chunk(delta: dict, finish: str | None = None) -> str:
627
+ return "data: " + _json.dumps({
628
+ "id": resp_id, "object": "chat.completion.chunk",
629
+ "created": created, "model": model, "session_id": session_id,
630
+ "choices": [{"index": 0, "delta": delta, "finish_reason": finish}],
631
+ }) + "\n\n"
632
+
633
+ async def _gen():
634
+ full_text = ""
635
+ t0 = time.monotonic()
636
+ yield _chunk({"role": "assistant"})
637
+ try:
638
+ cached = (_cache.get(user_content, session_id=session_id)
639
+ if settings.cache.enabled and user_content else None)
640
+ if cached:
641
+ full_text = cached.response
642
+ yield _chunk({"content": full_text})
643
+ else:
644
+ async for piece in provider.chat_stream(
645
+ messages=messages, model=model,
646
+ max_tokens=req.max_tokens or 4096,
647
+ **_sampling_kwargs(req),
648
+ ):
649
+ full_text += piece
650
+ yield _chunk({"content": piece})
651
+ except ProviderError as e:
652
+ # Mid-stream failure: SSE can't change the status code anymore —
653
+ # emit an explicit error event instead of silently truncating.
654
+ yield "data: " + _json.dumps({"error": {"message": str(e), "type": "provider_error"}}) + "\n\n"
655
+ yield _chunk({}, finish="stop")
656
+ yield "data: [DONE]\n\n"
657
+
658
+ # Post-stream bookkeeping
659
+ latency_ms = (time.monotonic() - t0) * 1000
660
+ output_tokens = count_tokens(full_text, model)
661
+ input_tokens = count_messages_tokens(messages, model)
662
+ if settings.cache.enabled and user_content and full_text:
663
+ _cache.set(user_content, full_text, input_tokens=input_tokens,
664
+ output_tokens=output_tokens, session_id=session_id)
665
+ _analytics.record(
666
+ session_id=session_id, provider=settings.provider, model=model,
667
+ input_tokens_original=orig_input_tokens,
668
+ input_tokens_sent=input_tokens, output_tokens=output_tokens,
669
+ tokens_saved=sum(savings.values()), latency_ms=latency_ms,
670
+ cache_hit=False, layer_savings=savings,
671
+ )
672
+
673
+ return StreamingResponse(_gen(), media_type="text/event-stream",
674
+ headers={"Cache-Control": "no-cache",
675
+ "X-Accel-Buffering": "no"})
676
+
677
+
612
678
  @app.post("/v1/chat/completions", dependencies=[Depends(verify_api_key), Depends(injection_guard)])
613
679
  async def chat_completions(req: ChatRequest, request: Request):
614
680
  """
@@ -662,6 +728,13 @@ async def chat_completions(req: ChatRequest, request: Request):
662
728
  session_id, graph, raw_messages, messages, model, savings, user_query
663
729
  )
664
730
 
731
+ # Streaming: true SSE passthrough (v0.3). Output-trimming is skipped in
732
+ # stream mode (can't trim tokens that already left the building) — all
733
+ # input-side layers (file intel, compression, graph context) still apply.
734
+ if req.stream:
735
+ return _stream_response(req, messages, model, user_content,
736
+ session_id, savings, orig_input_tokens)
737
+
665
738
  # Layer 5: call provider (or return cache hit)
666
739
  response_text, input_tokens_actual, output_tokens, latency_ms = await _call_provider(
667
740
  req, messages, model, user_content, session_id, savings
@@ -107,6 +107,21 @@ class BaseProvider(ABC):
107
107
 
108
108
  raise ProviderError(self.__class__.__name__, "max_retries", "All retry attempts exhausted", retryable=False)
109
109
 
110
+ def chat_stream(self, messages: list[dict], model: str = "",
111
+ max_tokens: int = 4096, system: str = "", **kwargs):
112
+ """Async generator yielding text chunks as the provider produces them.
113
+
114
+ Providers that support true streaming override this. The base
115
+ implementation raises so the API layer can return a clear 501 for
116
+ providers where passthrough streaming isn't implemented yet, instead
117
+ of silently degrading to a fake (buffered) stream.
118
+ """
119
+ raise ProviderError(
120
+ self.__class__.__name__, "stream_not_supported",
121
+ f"Streaming passthrough not implemented for {self.__class__.__name__}",
122
+ retryable=False,
123
+ )
124
+
110
125
 
111
126
  # ── Anthropic ─────────────────────────────────────────────────────────────────
112
127
 
@@ -179,6 +194,37 @@ class AnthropicProvider(BaseProvider):
179
194
  retryable = e.status_code in (500, 502, 503, 529)
180
195
  raise ProviderError("anthropic", f"http_{e.status_code}", str(e), retryable=retryable)
181
196
 
197
+ async def chat_stream(self, messages: list[dict], model: str = "",
198
+ max_tokens: int = 4096, system: str = "", **kwargs):
199
+ """True SSE passthrough — yields text chunks as Anthropic streams them."""
200
+ try:
201
+ import anthropic
202
+ except ImportError:
203
+ raise ImportError("pip install anthropic")
204
+ model = model or self.default_model
205
+ client = anthropic.AsyncAnthropic(api_key=self.api_key)
206
+
207
+ sys_parts = [m["content"] for m in messages if m.get("role") == "system"]
208
+ if system:
209
+ sys_parts.insert(0, system)
210
+ conv = [m for m in messages if m.get("role") != "system"]
211
+ kwargs_clean = _sampling(kwargs)
212
+ if "stop" in kwargs_clean:
213
+ kwargs_clean["stop_sequences"] = _as_stop_list(kwargs_clean.pop("stop"))
214
+ if sys_parts:
215
+ kwargs_clean["system"] = "\n\n".join(sys_parts)
216
+
217
+ try:
218
+ async with client.messages.stream(
219
+ model=model, messages=conv, max_tokens=max_tokens, **kwargs_clean
220
+ ) as s:
221
+ async for chunk in s.text_stream:
222
+ yield chunk
223
+ except anthropic.RateLimitError as e:
224
+ raise ProviderError("anthropic", "rate_limit", str(e), retryable=True)
225
+ except anthropic.APIStatusError as e:
226
+ raise ProviderError("anthropic", f"http_{e.status_code}", str(e), retryable=False)
227
+
182
228
 
183
229
  # ── OpenAI ────────────────────────────────────────────────────────────────────
184
230
 
@@ -237,6 +283,37 @@ class OpenAIProvider(BaseProvider):
237
283
  retryable = "rate" in err or "server" in err or "timeout" in err
238
284
  raise ProviderError("openai", "api_error", str(e), retryable=retryable)
239
285
 
286
+ async def chat_stream(self, messages: list[dict], model: str = "",
287
+ max_tokens: int = 4096, system: str = "", **kwargs):
288
+ """True SSE passthrough for OpenAI and all OpenAI-compatible providers
289
+ (DeepSeek, Mistral, OpenRouter, Grok inherit this)."""
290
+ try:
291
+ from openai import AsyncOpenAI
292
+ except ImportError:
293
+ raise ImportError("pip install openai")
294
+ model = model or self.default_model
295
+ client = AsyncOpenAI(
296
+ api_key=self.api_key,
297
+ **({"base_url": self._base_url} if self._base_url else {}),
298
+ )
299
+ all_messages = messages[:]
300
+ if system:
301
+ all_messages = [{"role": "system", "content": system}] + all_messages
302
+ try:
303
+ stream = await client.chat.completions.create(
304
+ model=model, messages=all_messages, max_tokens=max_tokens,
305
+ stream=True, **_sampling(kwargs),
306
+ )
307
+ async for chunk in stream:
308
+ delta = chunk.choices[0].delta.content if chunk.choices else None
309
+ if delta:
310
+ yield delta
311
+ except Exception as e:
312
+ err = str(e).lower()
313
+ retryable = "rate" in err or "server" in err or "timeout" in err
314
+ raise ProviderError(self.__class__.__name__.lower().replace("provider", ""),
315
+ "api_error", str(e), retryable=retryable)
316
+
240
317
 
241
318
  # ── DeepSeek ──────────────────────────────────────────────────────────────────
242
319
 
@@ -427,6 +504,37 @@ class OllamaProvider(BaseProvider):
427
504
  except Exception as e:
428
505
  raise ProviderError("ollama", "api_error", str(e), retryable=True)
429
506
 
507
+ async def chat_stream(self, messages: list[dict], model: str = "",
508
+ max_tokens: int = 4096, system: str = "", **kwargs):
509
+ """Streaming passthrough for local Ollama models."""
510
+ import json as _json
511
+
512
+ import httpx
513
+ model = model or self.default_model
514
+ all_messages = messages[:]
515
+ if system:
516
+ all_messages = [{"role": "system", "content": system}] + all_messages
517
+ payload = {"model": model, "messages": all_messages, "stream": True,
518
+ "options": {"num_predict": max_tokens}}
519
+ try:
520
+ async with httpx.AsyncClient(timeout=120) as client:
521
+ async with client.stream("POST", f"{self._base_url}/api/chat",
522
+ json=payload) as r:
523
+ r.raise_for_status()
524
+ async for line in r.aiter_lines():
525
+ if not line.strip():
526
+ continue
527
+ data = _json.loads(line)
528
+ chunk = data.get("message", {}).get("content", "")
529
+ if chunk:
530
+ yield chunk
531
+ if data.get("done"):
532
+ break
533
+ except ProviderError:
534
+ raise
535
+ except Exception as e:
536
+ raise ProviderError("ollama", "api_error", str(e), retryable=True)
537
+
430
538
 
431
539
  # ── Registry ──────────────────────────────────────────────────────────────────
432
540
 
@@ -1,29 +0,0 @@
1
- {
2
- "$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json",
3
- "name": "io.github.Shweta-Mishra-ai/tokenmizer",
4
- "description": "An MCP server that provides [describe what your server does]",
5
- "repository": {
6
- "url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
7
- "source": "github"
8
- },
9
- "version": "1.0.0",
10
- "packages": [
11
- {
12
- "registryType": "pypi",
13
- "identifier": "tokenmizer\"\r",
14
- "version": "1.0.0",
15
- "transport": {
16
- "type": "stdio"
17
- },
18
- "environmentVariables": [
19
- {
20
- "description": "Your API key for the service",
21
- "isRequired": true,
22
- "format": "string",
23
- "isSecret": true,
24
- "name": "YOUR_API_KEY"
25
- }
26
- ]
27
- }
28
- ]
29
- }
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes