tokenmizer 0.3.2__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/marketplace.json +1 -1
  2. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/plugin.json +1 -1
  3. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/workflows/ci.yml +5 -9
  4. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/CHANGELOG.md +39 -0
  5. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/PKG-INFO +34 -23
  6. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/README.md +33 -22
  7. tokenmizer-0.4.0/glama.json +6 -0
  8. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/pyproject.toml +1 -1
  9. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/mcp_e2e_check.py +16 -3
  10. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/server.json +2 -2
  11. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/test_api_endpoint.py +6 -9
  12. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_decision_tracker.py +10 -14
  13. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_graph.py +4 -5
  14. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_hybrid_extractor.py +7 -13
  15. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_mcp_server.py +15 -18
  16. tokenmizer-0.4.0/tests/unit/test_reasoning.py +244 -0
  17. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_security.py +1 -1
  18. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_validator.py +5 -6
  19. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_version_consistency.py +14 -7
  20. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/__init__.py +1 -1
  21. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/app.py +38 -14
  22. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/cli.py +3 -5
  23. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/engine.py +3 -5
  24. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/filters/file_intelligence.py +5 -6
  25. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/decision_tracker.py +23 -36
  26. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/graph.py +20 -28
  27. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/hybrid_extractor.py +8 -14
  28. tokenmizer-0.4.0/tokenmizer/graph_memory/ontology.py +143 -0
  29. tokenmizer-0.4.0/tokenmizer/graph_memory/reasoning.py +255 -0
  30. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/validator.py +7 -13
  31. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/visualization.py +7 -13
  32. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/mcp/server.py +76 -18
  33. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/redaction.py +7 -11
  34. tokenmizer-0.3.2/docs/DEMO_SCRIPT.md +0 -142
  35. tokenmizer-0.3.2/docs/superpowers/plans/2026-07-10-audit-fixes.md +0 -90
  36. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/analyze/SKILL.md +0 -0
  37. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/checkpoint/SKILL.md +0 -0
  38. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/resume/SKILL.md +0 -0
  39. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/stats/SKILL.md +0 -0
  40. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
  41. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/ISSUE_TEMPLATE/extraction_miss.md +0 -0
  42. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
  43. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/workflows/release.yml +0 -0
  44. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.gitignore +0 -0
  45. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.mcp.json +0 -0
  46. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/CONTRIBUTING.md +0 -0
  47. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/Dockerfile +0 -0
  48. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/LICENSE +0 -0
  49. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/SECURITY.md +0 -0
  50. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/TESTING.md +0 -0
  51. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/USAGE.md +0 -0
  52. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/__init__.py +0 -0
  53. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/__init__.py +0 -0
  54. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner.py +0 -0
  55. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner_v2.py +0 -0
  56. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner_v3.py +0 -0
  57. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/graph_retrieval/__init__.py +0 -0
  58. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/graph_retrieval/runner.py +0 -0
  59. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/latency/__init__.py +0 -0
  60. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/latency/runner.py +0 -0
  61. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/resume_quality/__init__.py +0 -0
  62. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docker-compose.yml +0 -0
  63. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/architecture.svg +0 -0
  64. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/demo.gif +0 -0
  65. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/logo.svg +0 -0
  66. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/examples/basic_usage.py +0 -0
  67. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/gen_demo_gif.py +0 -0
  68. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/install.sh +0 -0
  69. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/run_stdlib_tests.py +0 -0
  70. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/setup.sh +0 -0
  71. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/static_audit.py +0 -0
  72. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/__init__.py +0 -0
  73. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/chaos/__init__.py +0 -0
  74. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/chaos/test_recovery.py +0 -0
  75. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/conftest.py +0 -0
  76. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/__init__.py +0 -0
  77. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/test_checkpoint.py +0 -0
  78. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/memory_accuracy/__init__.py +0 -0
  79. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/memory_accuracy/test_retention.py +0 -0
  80. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/__init__.py +0 -0
  81. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_cache.py +0 -0
  82. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_compression_correctness.py +0 -0
  83. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_decision_cache_async.py +0 -0
  84. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_file_intelligence.py +0 -0
  85. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_graph_persistence.py +0 -0
  86. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_rate_limiter.py +0 -0
  87. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_tokenizer.py +0 -0
  88. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/agents/__init__.py +0 -0
  89. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/analytics/__init__.py +0 -0
  90. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/analytics/engine.py +0 -0
  91. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/__init__.py +0 -0
  92. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/rate_limiter.py +0 -0
  93. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/checkpoints/__init__.py +0 -0
  94. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/checkpoints/manager.py +0 -0
  95. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/__init__.py +0 -0
  96. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/output_trimmer.py +0 -0
  97. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/window.py +0 -0
  98. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/config/__init__.py +0 -0
  99. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/config/settings.py +0 -0
  100. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/__init__.py +0 -0
  101. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/dto.py +0 -0
  102. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/errors.py +0 -0
  103. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/tokenizer.py +0 -0
  104. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/dashboard/__init__.py +0 -0
  105. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/dashboard/page.py +0 -0
  106. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/filters/__init__.py +0 -0
  107. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/__init__.py +0 -0
  108. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/helpers.py +0 -0
  109. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/types.py +0 -0
  110. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/mcp/__init__.py +0 -0
  111. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/providers/__init__.py +0 -0
  112. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/providers/providers.py +0 -0
  113. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/__init__.py +0 -0
  114. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/auth.py +0 -0
  115. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/middleware.py +0 -0
  116. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/semantic_cache/__init__.py +0 -0
  117. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/semantic_cache/cache.py +0 -0
  118. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/state/__init__.py +0 -0
  119. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/state/backend.py +0 -0
  120. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/storage/__init__.py +0 -0
  121. {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer.yaml +0 -0
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tokenmizer",
3
- "version": "0.2.3",
3
+ "version": "0.4.0",
4
4
  "description": "Never lose your AI context again. Graph-backed memory, session checkpointing, and file intelligence for any LLM.",
5
5
  "homepage": "https://github.com/Shweta-Mishra-ai/tokenmizer",
6
6
  "repository": "https://github.com/Shweta-Mishra-ai/tokenmizer",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tokenmizer",
3
- "version": "0.3.2",
3
+ "version": "0.4.0",
4
4
  "description": "Never lose your AI context again. Graph-backed memory, session checkpointing, and file intelligence for Claude Code and any LLM.",
5
5
  "author": {
6
6
  "name": "Shweta Mishra",
@@ -8,13 +8,9 @@ on:
8
8
 
9
9
  jobs:
10
10
  test:
11
- # Windows is in the matrix because the 2026-07-10 audit found THREE
12
- # Windows-only bugs Linux CI could never catch: a cp1252
13
- # UnicodeEncodeError crashing `tokenmizer --help`, a leaked SQLite
14
- # handle blocking corrupt-DB recovery (WinError 32), and console
15
- # encoding issues in scripts. One Windows leg (latest Python only —
16
- # Windows runners are slow) keeps that whole bug class from ever
17
- # shipping again.
11
+ # Windows is in the matrix to catch platform-specific regressions
12
+ # (console encoding, file-handle semantics) that Linux legs cannot.
13
+ # One leg on the latest Python only — Windows runners are slow.
18
14
  runs-on: ${{ matrix.os }}
19
15
  strategy:
20
16
  fail-fast: false
@@ -47,8 +43,8 @@ jobs:
47
43
  - name: Run tests with coverage
48
44
  run: pytest tests/ --cov=tokenmizer --cov-report=term-missing --cov-report=xml -v
49
45
 
50
- # Regression guard for the cp1252 crash: --help must work on a
51
- # non-UTF-8 Windows console, not just under pytest's captured IO.
46
+ # --help must work on a non-UTF-8 Windows console, not just under
47
+ # pytest's captured IO (cp1252 cannot encode the CLI's emoji output).
52
48
  - name: CLI smoke test (Windows encoding regression)
53
49
  if: runner.os == 'Windows'
54
50
  run: python -m tokenmizer.cli --help
@@ -1,5 +1,44 @@
1
1
  # Changelog
2
2
 
3
+ ## [0.4.0] — 2026-07-11 — from storage to reasoning: ontology + graph reasoning
4
+
5
+ ### New — TokenMizer Ontology
6
+ - `tokenmizer/graph_memory/ontology.py`: the formal, machine-readable
7
+ vocabulary of the graph — every node type with semantics, every edge
8
+ type with domain/range/semantics, and the status **state machine**
9
+ (which lifecycle transitions are legal, e.g. COMPLETED→SUPERSEDED→
10
+ ARCHIVED; SUPERSEDED can never silently become COMPLETED again).
11
+ - Served at `GET /api/ontology` for MCP clients, docs, and tooling.
12
+ - Design principle: the ontology describes and audits, it does not gate
13
+ writes — graph ingestion stays permissive; violations are surfaced by
14
+ the consistency audit instead of causing silent data loss.
15
+
16
+ ### New — Graph Reasoning (`tokenmizer/graph_memory/reasoning.py`)
17
+ - **`why()`** — "Why is X the current choice?" Walks the supersession
18
+ chain in both directions and returns the old→new trail with trigger,
19
+ reason, and evidence per hop, plus the currently active decision.
20
+ `GET /api/graph/{id}/why?q=react`
21
+ - **`impact()`** — typed 1-hop neighborhood: which files/tasks/errors
22
+ connect to a node and via which relation.
23
+ - **`decision_history()`** — decision timeline grouped by topic bucket.
24
+ - **`consistency_check()`** — ontology-based audit: two active decisions
25
+ sharing a topic (contradictions the tracker missed), SUPERSEDED
26
+ decisions with no transition record (lost history), transitions
27
+ referencing pruned nodes.
28
+ - **`GET /api/graph/{id}/reasoning`** — the combined reasoning view.
29
+ - All reasoning is deterministic and local — no LLM calls.
30
+
31
+ ### New — MCP tool `why_decision` (6 tools now)
32
+ - Ask your agent "why did we pick X?" — it traces the decision trail:
33
+ struck-through old choices, replaced-by hops with reasons/evidence,
34
+ and the current active choice. Covered in unit tests and the e2e check.
35
+
36
+ ### Changed
37
+ - `glama.json` added (Glama MCP directory maintainer verification).
38
+ - README: "From Storage to Reasoning" section; internal demo-script and
39
+ planning docs removed from the repository.
40
+ - Version 0.4.0 everywhere (enforced by test_version_consistency).
41
+
3
42
  ## [0.3.2] — 2026-07-10 — full-repo audit: graph memory, MCP server, visualization
4
43
 
5
44
  ### Critical — the LLM extraction path never worked
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tokenmizer
3
- Version: 0.3.2
3
+ Version: 0.4.0
4
4
  Summary: Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session.
5
5
  Project-URL: Homepage, https://github.com/Shweta-Mishra-ai/tokenmizer
6
6
  Project-URL: Repository, https://github.com/Shweta-Mishra-ai/tokenmizer
@@ -153,13 +153,22 @@ Your App → TokenMizer (:8000) → Claude / GPT / Gemini / any LLM
153
153
  | 🔴 `INVALIDATED` | Explicitly wrong/cancelled | ⚠️ Always (warning) |
154
154
  | ⬜ `ARCHIVED` | Superseded >7 days ago — aged out | ❌ Never |
155
155
 
156
- History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable.
156
+ History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable:
157
+ ask `GET /api/graph/{session}/why?q=react` (or the `why_decision` MCP tool) and get the full
158
+ old → new trail with trigger, reason, and evidence per hop.
157
159
 
158
- > Honesty note (2026-07-10 audit): before v0.3.2, `ARCHIVED` was advertised
159
- > here but **unreachable** — the decay/prune/query logic all handled it, yet
160
- > no code path ever set it. Superseded decisions now age into `ARCHIVED`
161
- > automatically after 7 days (`GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS`),
162
- > which is what makes the "⚠️ 7 days" row above actually true.
160
+ ### From Storage to Reasoning
161
+
162
+ The graph doesn't just store facts — it answers questions over them:
163
+
164
+ | Capability | Endpoint / Tool | What it answers |
165
+ |---|---|---|
166
+ | **Ontology** | `GET /api/ontology` | The formal vocabulary: node/edge types with semantics, and the status state machine (which lifecycle transitions are legal) |
167
+ | **Causal chains** | `GET /api/graph/{id}/why?q=...` · MCP `why_decision` | "Why is X the current choice?" — walks the supersession chain with trigger/reason/evidence per hop |
168
+ | **Reasoning view** | `GET /api/graph/{id}/reasoning` | Active decisions per topic, recent changes, decision timeline, and a consistency audit |
169
+ | **Consistency audit** | (part of `/reasoning`) | Contradictions the tracker missed, superseded decisions with lost history, dangling references |
170
+
171
+ All reasoning is deterministic and local — no LLM calls, no extra cost.
163
172
 
164
173
  ---
165
174
 
@@ -360,10 +369,15 @@ env = { TOKENMIZER_URL = "http://localhost:8000" }
360
369
  </details>
361
370
 
362
371
  Then restart the client. Keep `tokenmizer serve` running for the
363
- checkpoint/resume/stats tools (file analysis works without it).
372
+ checkpoint/resume/stats/reasoning tools (file analysis works without it).
364
373
  If `tokenmizer-mcp` isn't on your PATH, use `"command": "python"`,
365
374
  `"args": ["-m", "tokenmizer.mcp.server"]` instead.
366
375
 
376
+ **Tools exposed (6):** `checkpoint_session`, `resume_session`,
377
+ `get_graph_stats`, `analyze_file`, `get_savings_stats`, and
378
+ `why_decision` — ask your agent *"why did we pick X?"* and it traces the
379
+ decision's supersession chain with reasons and evidence.
380
+
367
381
  ---
368
382
 
369
383
  ## Other Tools
@@ -461,12 +475,8 @@ graph_checkpoint:
461
475
  enabled: true
462
476
  trigger_at_percent: 0.85
463
477
  use_llm_extraction: false # true = hybrid LLM+heuristic extraction
464
- # (needs a provider key, ~$0.001/turn).
465
- # NOTE: before v0.3.2 this flag silently did
466
- # nothing — the call site passed provider_fn
467
- # to the wrong function and raised TypeError
468
- # on every turn, falling back to heuristics.
469
- # Fixed + regression-tested; see CHANGELOG.
478
+ # (needs a provider key, ~$0.001/turn;
479
+ # requires v0.3.2+ — see CHANGELOG)
470
480
 
471
481
  compression:
472
482
  enabled: true
@@ -507,6 +517,9 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
507
517
  | `/api/decision/invalidate` | POST | Mark decision as invalid |
508
518
  | `/api/graph/{id}` | GET | Session graph stats |
509
519
  | `/api/graph/{id}/html` | GET | **Interactive graph page** — decision-history timeline, supersession arcs, type/status filters, search, zoom/pan, PNG export. Zero external dependencies (works offline) |
520
+ | `/api/graph/{id}/why?q=` | GET | **Reasoning:** causal chain behind a decision (old → new with trigger/reason/evidence) |
521
+ | `/api/graph/{id}/reasoning` | GET | **Reasoning view:** active decisions by topic, recent changes, consistency audit |
522
+ | `/api/ontology` | GET | Machine-readable graph ontology (types, relations, status state machine) |
510
523
  | `/api/stats` | GET | Token savings analytics |
511
524
  | `/health` | GET | Health check |
512
525
  | `/docs` | GET | Swagger UI |
@@ -517,15 +530,13 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
517
530
 
518
531
  - API key auth — `TOKENMIZER_API_KEY` (constant-time comparison)
519
532
  - Secret/PII redaction applied once at ingestion, before graph storage,
520
- checkpoint storage, AND every LLM call (main chat *and* the background
521
- extraction model — these are separate, the redaction gap between them
522
- was a real bug, now fixed). Patterns cover Anthropic/OpenAI/Google/
523
- GitHub/AWS/Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded
524
- credentials (`postgres://user:pass@host` — a 2026-07 audit gap, fixed),
525
- and generic `key=`/`password=` assignments. Best-effort by nature —
526
- an unrecognized format with no keyword context can still slip through.
527
- The checkpoint layer independently re-redacts what it persists
528
- (defense-in-depth, added in the same audit).
533
+ checkpoint storage, and every LLM call (main chat and the background
534
+ extraction model). Patterns cover Anthropic/OpenAI/Google/GitHub/AWS/
535
+ Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded credentials
536
+ (`postgres://user:pass@host`), and generic `key=`/`password=`
537
+ assignments. Best-effort by nature — an unrecognized format with no
538
+ keyword context can still slip through. The checkpoint layer
539
+ independently re-redacts what it persists (defense in depth).
529
540
  - Session-isolated cache (sensitive data never shared across sessions)
530
541
  - Basic prompt-injection keyword filter — catches copy-pasted jailbreak
531
542
  templates only; **not** a security boundary against a motivated
@@ -78,13 +78,22 @@ Your App → TokenMizer (:8000) → Claude / GPT / Gemini / any LLM
78
78
  | 🔴 `INVALIDATED` | Explicitly wrong/cancelled | ⚠️ Always (warning) |
79
79
  | ⬜ `ARCHIVED` | Superseded >7 days ago — aged out | ❌ Never |
80
80
 
81
- History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable.
81
+ History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable:
82
+ ask `GET /api/graph/{session}/why?q=react` (or the `why_decision` MCP tool) and get the full
83
+ old → new trail with trigger, reason, and evidence per hop.
82
84
 
83
- > Honesty note (2026-07-10 audit): before v0.3.2, `ARCHIVED` was advertised
84
- > here but **unreachable** — the decay/prune/query logic all handled it, yet
85
- > no code path ever set it. Superseded decisions now age into `ARCHIVED`
86
- > automatically after 7 days (`GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS`),
87
- > which is what makes the "⚠️ 7 days" row above actually true.
85
+ ### From Storage to Reasoning
86
+
87
+ The graph doesn't just store facts — it answers questions over them:
88
+
89
+ | Capability | Endpoint / Tool | What it answers |
90
+ |---|---|---|
91
+ | **Ontology** | `GET /api/ontology` | The formal vocabulary: node/edge types with semantics, and the status state machine (which lifecycle transitions are legal) |
92
+ | **Causal chains** | `GET /api/graph/{id}/why?q=...` · MCP `why_decision` | "Why is X the current choice?" — walks the supersession chain with trigger/reason/evidence per hop |
93
+ | **Reasoning view** | `GET /api/graph/{id}/reasoning` | Active decisions per topic, recent changes, decision timeline, and a consistency audit |
94
+ | **Consistency audit** | (part of `/reasoning`) | Contradictions the tracker missed, superseded decisions with lost history, dangling references |
95
+
96
+ All reasoning is deterministic and local — no LLM calls, no extra cost.
88
97
 
89
98
  ---
90
99
 
@@ -285,10 +294,15 @@ env = { TOKENMIZER_URL = "http://localhost:8000" }
285
294
  </details>
286
295
 
287
296
  Then restart the client. Keep `tokenmizer serve` running for the
288
- checkpoint/resume/stats tools (file analysis works without it).
297
+ checkpoint/resume/stats/reasoning tools (file analysis works without it).
289
298
  If `tokenmizer-mcp` isn't on your PATH, use `"command": "python"`,
290
299
  `"args": ["-m", "tokenmizer.mcp.server"]` instead.
291
300
 
301
+ **Tools exposed (6):** `checkpoint_session`, `resume_session`,
302
+ `get_graph_stats`, `analyze_file`, `get_savings_stats`, and
303
+ `why_decision` — ask your agent *"why did we pick X?"* and it traces the
304
+ decision's supersession chain with reasons and evidence.
305
+
292
306
  ---
293
307
 
294
308
  ## Other Tools
@@ -386,12 +400,8 @@ graph_checkpoint:
386
400
  enabled: true
387
401
  trigger_at_percent: 0.85
388
402
  use_llm_extraction: false # true = hybrid LLM+heuristic extraction
389
- # (needs a provider key, ~$0.001/turn).
390
- # NOTE: before v0.3.2 this flag silently did
391
- # nothing — the call site passed provider_fn
392
- # to the wrong function and raised TypeError
393
- # on every turn, falling back to heuristics.
394
- # Fixed + regression-tested; see CHANGELOG.
403
+ # (needs a provider key, ~$0.001/turn;
404
+ # requires v0.3.2+ — see CHANGELOG)
395
405
 
396
406
  compression:
397
407
  enabled: true
@@ -432,6 +442,9 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
432
442
  | `/api/decision/invalidate` | POST | Mark decision as invalid |
433
443
  | `/api/graph/{id}` | GET | Session graph stats |
434
444
  | `/api/graph/{id}/html` | GET | **Interactive graph page** — decision-history timeline, supersession arcs, type/status filters, search, zoom/pan, PNG export. Zero external dependencies (works offline) |
445
+ | `/api/graph/{id}/why?q=` | GET | **Reasoning:** causal chain behind a decision (old → new with trigger/reason/evidence) |
446
+ | `/api/graph/{id}/reasoning` | GET | **Reasoning view:** active decisions by topic, recent changes, consistency audit |
447
+ | `/api/ontology` | GET | Machine-readable graph ontology (types, relations, status state machine) |
435
448
  | `/api/stats` | GET | Token savings analytics |
436
449
  | `/health` | GET | Health check |
437
450
  | `/docs` | GET | Swagger UI |
@@ -442,15 +455,13 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
442
455
 
443
456
  - API key auth — `TOKENMIZER_API_KEY` (constant-time comparison)
444
457
  - Secret/PII redaction applied once at ingestion, before graph storage,
445
- checkpoint storage, AND every LLM call (main chat *and* the background
446
- extraction model — these are separate, the redaction gap between them
447
- was a real bug, now fixed). Patterns cover Anthropic/OpenAI/Google/
448
- GitHub/AWS/Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded
449
- credentials (`postgres://user:pass@host` — a 2026-07 audit gap, fixed),
450
- and generic `key=`/`password=` assignments. Best-effort by nature —
451
- an unrecognized format with no keyword context can still slip through.
452
- The checkpoint layer independently re-redacts what it persists
453
- (defense-in-depth, added in the same audit).
458
+ checkpoint storage, and every LLM call (main chat and the background
459
+ extraction model). Patterns cover Anthropic/OpenAI/Google/GitHub/AWS/
460
+ Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded credentials
461
+ (`postgres://user:pass@host`), and generic `key=`/`password=`
462
+ assignments. Best-effort by nature — an unrecognized format with no
463
+ keyword context can still slip through. The checkpoint layer
464
+ independently re-redacts what it persists (defense in depth).
454
465
  - Session-isolated cache (sensitive data never shared across sessions)
455
466
  - Basic prompt-injection keyword filter — catches copy-pasted jailbreak
456
467
  templates only; **not** a security boundary against a motivated
@@ -0,0 +1,6 @@
1
+ {
2
+ "$schema": "https://glama.ai/mcp/schemas/server.json",
3
+ "maintainers": [
4
+ "Shweta-Mishra-ai"
5
+ ]
6
+ }
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "tokenmizer"
7
- version = "0.3.2"
7
+ version = "0.4.0"
8
8
  description = "Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -108,8 +108,8 @@ def main() -> int:
108
108
  r = rpc("tools/list", req_id=2)
109
109
  tools = {t["name"] for t in r.get("result", {}).get("tools", [])}
110
110
  expected = {"checkpoint_session", "resume_session", "get_graph_stats",
111
- "analyze_file", "get_savings_stats"}
112
- check("tools/list exposes all 5 tools", tools == expected, str(tools))
111
+ "analyze_file", "get_savings_stats", "why_decision"}
112
+ check("tools/list exposes all 6 tools", tools == expected, str(tools))
113
113
 
114
114
  # 3. Local tool (no proxy): analyze_file on a real CSV
115
115
  with tempfile.NamedTemporaryFile("w", suffix=".csv", delete=False,
@@ -137,8 +137,21 @@ def main() -> int:
137
137
  text = r.get("result", {}).get("content", [{}])[0].get("text", "")
138
138
  check("resume_session returns context", "TokenMizer Resume" in text, text[:160])
139
139
 
140
+ # Reasoning tool round-trips through the proxy. Whether the test
141
+ # session happens to contain a matching decision or not, a healthy
142
+ # server answers with a trail or a clean "no match" — never an error.
143
+ r = rpc("tools/call", {"name": "why_decision",
144
+ "arguments": {"session_id": "mcp-e2e-test",
145
+ "query": "postgres"}}, req_id=7)
146
+ res = r.get("result", {})
147
+ text = res.get("content", [{}])[0].get("text", "")
148
+ check("why_decision answers without error",
149
+ res.get("isError") is False
150
+ and ("Decision trail" in text or "No decision matching" in text),
151
+ text[:160])
152
+
140
153
  # 5. Unknown method → JSON-RPC error, not crash
141
- r = rpc("bogus/method", {}, req_id=7)
154
+ r = rpc("bogus/method", {}, req_id=8)
142
155
  check("unknown method returns -32601 error",
143
156
  r.get("error", {}).get("code") == -32601)
144
157
 
@@ -6,12 +6,12 @@
6
6
  "url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
7
7
  "source": "github"
8
8
  },
9
- "version": "0.3.2",
9
+ "version": "0.4.0",
10
10
  "packages": [
11
11
  {
12
12
  "registryType": "pypi",
13
13
  "identifier": "tokenmizer",
14
- "version": "0.3.2",
14
+ "version": "0.4.0",
15
15
  "transport": { "type": "stdio" },
16
16
  "environmentVariables": [
17
17
  {
@@ -194,10 +194,8 @@ class TestHealthAndDocs:
194
194
  def test_graph_share_html(self, client):
195
195
  """Shareable graph page: self-contained interactive HTML.
196
196
 
197
- REDESIGNED (2026-07-10): no more D3-from-CDN — the artifact must
198
- work offline. It must also actually render the supersession
199
- history (transitions), which the old template computed and then
200
- silently never used.
197
+ The artifact must work offline (no CDN/external loads) and must
198
+ embed the supersession history (transitions) it renders.
201
199
  """
202
200
  c, _ = client
203
201
  # Put something in the graph via the pipeline first
@@ -223,11 +221,10 @@ class TestHealthAndDocs:
223
221
 
224
222
  class TestInvalidateDecisionScope:
225
223
  """
226
- AUDIT FIX (2026-07-10): /api/decision/invalidate previously substring-
227
- matched across ALL decision nodes regardless of status — invalidating
228
- "postgres" would also flip an already-SUPERSEDED postgres decision,
229
- destroying its supersession history. Only ACTIVE (COMPLETED) decisions
230
- are now eligible, and the response lists exactly what was affected.
224
+ /api/decision/invalidate must only match ACTIVE (COMPLETED) decisions:
225
+ matching across all statuses would overwrite SUPERSEDED history nodes
226
+ and destroy their supersession record. The response must list every
227
+ affected node.
231
228
  """
232
229
 
233
230
  def _seed_graph(self, session_id, tmp_path):
@@ -1,14 +1,10 @@
1
1
  """
2
- Regression tests for the topic classifier (2026-07-10 audit).
3
-
4
- Confirmed misclassifications before the fix (all reproduced by execution):
5
- - "Go with tRPC for the API layer" → "language" (imperative "Go" collided
6
- with Go-the-language; first-single-word-hit-wins never reached "trpc")
7
- - "Use Supabase for the backend" → None (vocabulary gap; the sibling
8
- hybrid_extractor.py knew "supabase" but this file didn't)
9
- - "Use Clerk for authentication" → None (same gap)
10
- - "Use FastAPI with SQLAlchemy and PostgreSQL" → only "web_framework";
11
- a later Postgres→SQLite switch was never detected as contradicting it.
2
+ Tests for the decision topic classifier and same-decision detection.
3
+
4
+ Covers the known-hard cases: the imperative "Go with X" vs. Go-the-language
5
+ ambiguity, technology names shared with the extractor vocabulary, multi-topic
6
+ statements, bigram/single-word precedence, and near-duplicate label
7
+ containment.
12
8
  """
13
9
  from tokenmizer.graph_memory.decision_tracker import (
14
10
  classify_topic,
@@ -112,12 +108,12 @@ def test_same_decision_not_superseded(tmp_path):
112
108
  assert old_id not in hits
113
109
 
114
110
 
115
- # ── AUDIT round 2 (2026-07-10): near-duplicate decision merging ──────────────
111
+ # ── Near-duplicate decision merging ──────────────────────────────────────────
116
112
 
117
113
  def test_containment_variant_merges_not_supersedes(tmp_path):
118
- """The demo bug: 'use React for the frontend.' and 'Use React' were
119
- emitted from ONE message, became two nodes, and one superseded the
120
- other — a bogus 'Changed:' line in every resume. They must merge."""
114
+ """Two label variants of one decision (emitted from a single message)
115
+ must merge into one node rather than supersede each other, which would
116
+ record a spurious decision change."""
121
117
  g = GraphMemory(session_id="t-dup", storage_dir=str(tmp_path))
122
118
  id1 = g.add_node(NodeType.DECISION, "use React for the frontend.",
123
119
  NodeStatus.COMPLETED)
@@ -203,11 +203,10 @@ class TestPersistence:
203
203
 
204
204
  class TestArchivedReachability:
205
205
  """
206
- AUDIT FIX (2026-07-10): ARCHIVED was a documented, fully-wired status
207
- (decay rates, prune rules, query filters all handled it) that NOTHING
208
- ever set — the README advertised a 4-state model whose 4th state was
209
- unreachable. SUPERSEDED decisions now age into ARCHIVED after
210
- GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS days (from supersession time).
206
+ SUPERSEDED decisions age into ARCHIVED after
207
+ GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS, measured from supersession
208
+ time. apply_importance_decay() is the only path that sets ARCHIVED,
209
+ so these tests guard the state's reachability.
211
210
  """
212
211
 
213
212
  def test_superseded_decision_ages_into_archived(self, graph):
@@ -151,15 +151,10 @@ async def test_extract_without_provider():
151
151
  @pytest.mark.asyncio
152
152
  async def test_extract_llm_pass_actually_invoked_app_style():
153
153
  """
154
- REGRESSION (2026-07-10 audit): api/app.py's background extraction did
155
- HybridExtractor(provider_fn=_pfn) → TypeError (no such kwarg), and
156
- ext.extract(_msgs) → provider_fn=None, LLM pass skipped.
157
- So the LLM extraction path NEVER ran in production — every call raised,
158
- was caught by the broad except, and logged as a provider failure.
159
-
160
- This test replicates the app.py call pattern exactly as fixed:
161
- construct with defaults, pass provider_fn to extract(), and assert
162
- the provider was actually called and its output merged in.
154
+ Regression guard for the api/app.py call pattern: construct with
155
+ defaults and pass provider_fn to extract(). Asserts the provider is
156
+ actually invoked and its output is merged — a call-signature drift
157
+ here silently disables the LLM pass.
163
158
  """
164
159
  calls = []
165
160
 
@@ -183,10 +178,9 @@ async def test_extract_llm_pass_actually_invoked_app_style():
183
178
 
184
179
  class TestMinConfidenceFilter:
185
180
  """
186
- AUDIT round 2 (2026-07-10): min_confidence was dead code — stored in
187
- __init__, never read. It now filters extract() output by merge()'s
188
- confidence tiers (0.95 corroborated / 0.80 LLM-only / 0.65 heuristic-
189
- only). Default 0.55 keeps everything (backward compatible).
181
+ min_confidence filters extract() output by merge()'s confidence tiers
182
+ (0.95 corroborated / 0.80 LLM-only / 0.65 heuristic-only). The default
183
+ of 0.55 keeps every tier.
190
184
  """
191
185
 
192
186
  def test_default_keeps_heuristic_only_items(self):
@@ -1,16 +1,13 @@
1
1
  """
2
- Regression tests for the MCP stdio server (2026-07-10 audit).
3
-
4
- Bugs these guard against (all confirmed live before the fix):
5
- 1. isError was computed via result_text.startswith("❌") — a KeyError from
6
- a missing required argument surfaced as "Tool error: 'session_id'" with
7
- isError FALSE, i.e. clients saw a *successful* result.
8
- 2. A valid-JSON-but-not-an-object line ([1,2]) crashed the whole server
9
- via AttributeError on req.get().
10
- 3. Any exception in a handler outside tools/call (e.g. initialize) killed
11
- the process with no JSON-RPC error — the client hung, then saw the
12
- subprocess die.
13
- 4. Malformed JSON lines were silently dropped (no log, no error response).
2
+ Regression tests for the MCP stdio server.
3
+
4
+ Invariants under test:
5
+ 1. isError is structural — validation failures and handler crashes are
6
+ reported with isError: true regardless of message text.
7
+ 2. No input terminates the read loop: malformed JSON returns -32700,
8
+ non-object messages return -32600, handler exceptions return -32603,
9
+ and subsequent requests are still served.
10
+ 3. Required arguments are validated with typed, descriptive errors.
14
11
  """
15
12
  import io
16
13
  import json
@@ -81,14 +78,14 @@ def test_analyze_file_success_not_error(tmp_path):
81
78
 
82
79
 
83
80
  def test_handler_crash_is_error_not_exception(monkeypatch):
84
- def boom(args):
85
- raise RuntimeError("kaboom")
81
+ def failing_handler(args):
82
+ raise RuntimeError("simulated handler failure")
86
83
  # handle_tool_call builds its dispatch dict from module globals at call
87
84
  # time, so patching the module attribute is picked up.
88
- monkeypatch.setattr(mcp, "handle_get_savings_stats", boom)
85
+ monkeypatch.setattr(mcp, "handle_get_savings_stats", failing_handler)
89
86
  text, is_error = mcp.handle_tool_call("get_savings_stats", {})
90
87
  assert is_error is True
91
- assert "kaboom" in text or "internal error" in text.lower()
88
+ assert "simulated handler failure" in text or "internal error" in text.lower()
92
89
 
93
90
 
94
91
  # ── stdio transport: survives hostile input ──────────────────────────────────
@@ -120,7 +117,7 @@ def test_stdio_survives_malformed_json(monkeypatch):
120
117
  ])
121
118
  assert out[0]["error"]["code"] == -32700
122
119
  assert out[1]["id"] == 2
123
- assert len(out[1]["result"]["tools"]) == 5
120
+ assert len(out[1]["result"]["tools"]) == 6
124
121
 
125
122
 
126
123
  def test_stdio_survives_non_object_json(monkeypatch):
@@ -159,7 +156,7 @@ def test_stdio_handler_exception_yields_jsonrpc_error(monkeypatch):
159
156
 
160
157
 
161
158
  def test_stdio_missing_arg_reports_is_error_true(monkeypatch):
162
- """THE bug: missing session_id must reach the client as isError: true."""
159
+ """A missing required argument must reach the client as isError: true."""
163
160
  out = _run_lines(monkeypatch, [
164
161
  json.dumps({"jsonrpc": "2.0", "id": 6, "method": "tools/call",
165
162
  "params": {"name": "checkpoint_session", "arguments": {}}}),