tokenmizer 0.3.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/marketplace.json +1 -1
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/plugin.json +1 -1
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/workflows/ci.yml +5 -9
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/CHANGELOG.md +39 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/PKG-INFO +34 -23
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/README.md +33 -22
- tokenmizer-0.4.0/glama.json +6 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/pyproject.toml +1 -1
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/mcp_e2e_check.py +16 -3
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/server.json +2 -2
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/test_api_endpoint.py +6 -9
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_decision_tracker.py +10 -14
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_graph.py +4 -5
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_hybrid_extractor.py +7 -13
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_mcp_server.py +15 -18
- tokenmizer-0.4.0/tests/unit/test_reasoning.py +244 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_security.py +1 -1
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_validator.py +5 -6
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_version_consistency.py +14 -7
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/__init__.py +1 -1
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/app.py +38 -14
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/cli.py +3 -5
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/engine.py +3 -5
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/filters/file_intelligence.py +5 -6
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/decision_tracker.py +23 -36
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/graph.py +20 -28
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/hybrid_extractor.py +8 -14
- tokenmizer-0.4.0/tokenmizer/graph_memory/ontology.py +143 -0
- tokenmizer-0.4.0/tokenmizer/graph_memory/reasoning.py +255 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/validator.py +7 -13
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/visualization.py +7 -13
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/mcp/server.py +76 -18
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/redaction.py +7 -11
- tokenmizer-0.3.2/docs/DEMO_SCRIPT.md +0 -142
- tokenmizer-0.3.2/docs/superpowers/plans/2026-07-10-audit-fixes.md +0 -90
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/analyze/SKILL.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/checkpoint/SKILL.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/resume/SKILL.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.claude-plugin/skills/stats/SKILL.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/ISSUE_TEMPLATE/bug_report.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/ISSUE_TEMPLATE/extraction_miss.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/PULL_REQUEST_TEMPLATE.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.github/workflows/release.yml +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.gitignore +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/.mcp.json +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/CONTRIBUTING.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/Dockerfile +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/LICENSE +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/SECURITY.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/TESTING.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/USAGE.md +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner_v2.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/checkpoint_accuracy/runner_v3.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/graph_retrieval/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/graph_retrieval/runner.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/latency/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/latency/runner.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/benchmarks/resume_quality/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docker-compose.yml +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/architecture.svg +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/demo.gif +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/docs/assets/logo.svg +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/examples/basic_usage.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/gen_demo_gif.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/install.sh +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/run_stdlib_tests.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/setup.sh +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/scripts/static_audit.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/chaos/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/chaos/test_recovery.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/conftest.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/integration/test_checkpoint.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/memory_accuracy/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/memory_accuracy/test_retention.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_cache.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_compression_correctness.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_decision_cache_async.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_file_intelligence.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_graph_persistence.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_rate_limiter.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tests/unit/test_tokenizer.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/agents/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/analytics/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/analytics/engine.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/api/rate_limiter.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/checkpoints/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/checkpoints/manager.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/output_trimmer.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/compression/window.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/config/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/config/settings.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/dto.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/errors.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/core/tokenizer.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/dashboard/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/dashboard/page.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/filters/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/helpers.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/graph_memory/types.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/mcp/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/providers/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/providers/providers.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/auth.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/security/middleware.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/semantic_cache/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/semantic_cache/cache.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/state/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/state/backend.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer/storage/__init__.py +0 -0
- {tokenmizer-0.3.2 → tokenmizer-0.4.0}/tokenmizer.yaml +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "tokenmizer",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.4.0",
|
|
4
4
|
"description": "Never lose your AI context again. Graph-backed memory, session checkpointing, and file intelligence for any LLM.",
|
|
5
5
|
"homepage": "https://github.com/Shweta-Mishra-ai/tokenmizer",
|
|
6
6
|
"repository": "https://github.com/Shweta-Mishra-ai/tokenmizer",
|
|
@@ -8,13 +8,9 @@ on:
|
|
|
8
8
|
|
|
9
9
|
jobs:
|
|
10
10
|
test:
|
|
11
|
-
# Windows is in the matrix
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
# handle blocking corrupt-DB recovery (WinError 32), and console
|
|
15
|
-
# encoding issues in scripts. One Windows leg (latest Python only —
|
|
16
|
-
# Windows runners are slow) keeps that whole bug class from ever
|
|
17
|
-
# shipping again.
|
|
11
|
+
# Windows is in the matrix to catch platform-specific regressions
|
|
12
|
+
# (console encoding, file-handle semantics) that Linux legs cannot.
|
|
13
|
+
# One leg on the latest Python only — Windows runners are slow.
|
|
18
14
|
runs-on: ${{ matrix.os }}
|
|
19
15
|
strategy:
|
|
20
16
|
fail-fast: false
|
|
@@ -47,8 +43,8 @@ jobs:
|
|
|
47
43
|
- name: Run tests with coverage
|
|
48
44
|
run: pytest tests/ --cov=tokenmizer --cov-report=term-missing --cov-report=xml -v
|
|
49
45
|
|
|
50
|
-
#
|
|
51
|
-
#
|
|
46
|
+
# --help must work on a non-UTF-8 Windows console, not just under
|
|
47
|
+
# pytest's captured IO (cp1252 cannot encode the CLI's emoji output).
|
|
52
48
|
- name: CLI smoke test (Windows encoding regression)
|
|
53
49
|
if: runner.os == 'Windows'
|
|
54
50
|
run: python -m tokenmizer.cli --help
|
|
@@ -1,5 +1,44 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## [0.4.0] — 2026-07-11 — from storage to reasoning: ontology + graph reasoning
|
|
4
|
+
|
|
5
|
+
### New — TokenMizer Ontology
|
|
6
|
+
- `tokenmizer/graph_memory/ontology.py`: the formal, machine-readable
|
|
7
|
+
vocabulary of the graph — every node type with semantics, every edge
|
|
8
|
+
type with domain/range/semantics, and the status **state machine**
|
|
9
|
+
(which lifecycle transitions are legal, e.g. COMPLETED→SUPERSEDED→
|
|
10
|
+
ARCHIVED; SUPERSEDED can never silently become COMPLETED again).
|
|
11
|
+
- Served at `GET /api/ontology` for MCP clients, docs, and tooling.
|
|
12
|
+
- Design principle: the ontology describes and audits, it does not gate
|
|
13
|
+
writes — graph ingestion stays permissive; violations are surfaced by
|
|
14
|
+
the consistency audit instead of causing silent data loss.
|
|
15
|
+
|
|
16
|
+
### New — Graph Reasoning (`tokenmizer/graph_memory/reasoning.py`)
|
|
17
|
+
- **`why()`** — "Why is X the current choice?" Walks the supersession
|
|
18
|
+
chain in both directions and returns the old→new trail with trigger,
|
|
19
|
+
reason, and evidence per hop, plus the currently active decision.
|
|
20
|
+
`GET /api/graph/{id}/why?q=react`
|
|
21
|
+
- **`impact()`** — typed 1-hop neighborhood: which files/tasks/errors
|
|
22
|
+
connect to a node and via which relation.
|
|
23
|
+
- **`decision_history()`** — decision timeline grouped by topic bucket.
|
|
24
|
+
- **`consistency_check()`** — ontology-based audit: two active decisions
|
|
25
|
+
sharing a topic (contradictions the tracker missed), SUPERSEDED
|
|
26
|
+
decisions with no transition record (lost history), transitions
|
|
27
|
+
referencing pruned nodes.
|
|
28
|
+
- **`GET /api/graph/{id}/reasoning`** — the combined reasoning view.
|
|
29
|
+
- All reasoning is deterministic and local — no LLM calls.
|
|
30
|
+
|
|
31
|
+
### New — MCP tool `why_decision` (6 tools now)
|
|
32
|
+
- Ask your agent "why did we pick X?" — it traces the decision trail:
|
|
33
|
+
struck-through old choices, replaced-by hops with reasons/evidence,
|
|
34
|
+
and the current active choice. Covered in unit tests and the e2e check.
|
|
35
|
+
|
|
36
|
+
### Changed
|
|
37
|
+
- `glama.json` added (Glama MCP directory maintainer verification).
|
|
38
|
+
- README: "From Storage to Reasoning" section; internal demo-script and
|
|
39
|
+
planning docs removed from the repository.
|
|
40
|
+
- Version 0.4.0 everywhere (enforced by test_version_consistency).
|
|
41
|
+
|
|
3
42
|
## [0.3.2] — 2026-07-10 — full-repo audit: graph memory, MCP server, visualization
|
|
4
43
|
|
|
5
44
|
### Critical — the LLM extraction path never worked
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenmizer
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Reduce AI context loss by 2x. Graph-backed checkpoint and resume for any LLM session.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
6
6
|
Project-URL: Repository, https://github.com/Shweta-Mishra-ai/tokenmizer
|
|
@@ -153,13 +153,22 @@ Your App → TokenMizer (:8000) → Claude / GPT / Gemini / any LLM
|
|
|
153
153
|
| 🔴 `INVALIDATED` | Explicitly wrong/cancelled | ⚠️ Always (warning) |
|
|
154
154
|
| ⬜ `ARCHIVED` | Superseded >7 days ago — aged out | ❌ Never |
|
|
155
155
|
|
|
156
|
-
History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable
|
|
156
|
+
History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable:
|
|
157
|
+
ask `GET /api/graph/{session}/why?q=react` (or the `why_decision` MCP tool) and get the full
|
|
158
|
+
old → new trail with trigger, reason, and evidence per hop.
|
|
157
159
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
160
|
+
### From Storage to Reasoning
|
|
161
|
+
|
|
162
|
+
The graph doesn't just store facts — it answers questions over them:
|
|
163
|
+
|
|
164
|
+
| Capability | Endpoint / Tool | What it answers |
|
|
165
|
+
|---|---|---|
|
|
166
|
+
| **Ontology** | `GET /api/ontology` | The formal vocabulary: node/edge types with semantics, and the status state machine (which lifecycle transitions are legal) |
|
|
167
|
+
| **Causal chains** | `GET /api/graph/{id}/why?q=...` · MCP `why_decision` | "Why is X the current choice?" — walks the supersession chain with trigger/reason/evidence per hop |
|
|
168
|
+
| **Reasoning view** | `GET /api/graph/{id}/reasoning` | Active decisions per topic, recent changes, decision timeline, and a consistency audit |
|
|
169
|
+
| **Consistency audit** | (part of `/reasoning`) | Contradictions the tracker missed, superseded decisions with lost history, dangling references |
|
|
170
|
+
|
|
171
|
+
All reasoning is deterministic and local — no LLM calls, no extra cost.
|
|
163
172
|
|
|
164
173
|
---
|
|
165
174
|
|
|
@@ -360,10 +369,15 @@ env = { TOKENMIZER_URL = "http://localhost:8000" }
|
|
|
360
369
|
</details>
|
|
361
370
|
|
|
362
371
|
Then restart the client. Keep `tokenmizer serve` running for the
|
|
363
|
-
checkpoint/resume/stats tools (file analysis works without it).
|
|
372
|
+
checkpoint/resume/stats/reasoning tools (file analysis works without it).
|
|
364
373
|
If `tokenmizer-mcp` isn't on your PATH, use `"command": "python"`,
|
|
365
374
|
`"args": ["-m", "tokenmizer.mcp.server"]` instead.
|
|
366
375
|
|
|
376
|
+
**Tools exposed (6):** `checkpoint_session`, `resume_session`,
|
|
377
|
+
`get_graph_stats`, `analyze_file`, `get_savings_stats`, and
|
|
378
|
+
`why_decision` — ask your agent *"why did we pick X?"* and it traces the
|
|
379
|
+
decision's supersession chain with reasons and evidence.
|
|
380
|
+
|
|
367
381
|
---
|
|
368
382
|
|
|
369
383
|
## Other Tools
|
|
@@ -461,12 +475,8 @@ graph_checkpoint:
|
|
|
461
475
|
enabled: true
|
|
462
476
|
trigger_at_percent: 0.85
|
|
463
477
|
use_llm_extraction: false # true = hybrid LLM+heuristic extraction
|
|
464
|
-
# (needs a provider key, ~$0.001/turn
|
|
465
|
-
#
|
|
466
|
-
# nothing — the call site passed provider_fn
|
|
467
|
-
# to the wrong function and raised TypeError
|
|
468
|
-
# on every turn, falling back to heuristics.
|
|
469
|
-
# Fixed + regression-tested; see CHANGELOG.
|
|
478
|
+
# (needs a provider key, ~$0.001/turn;
|
|
479
|
+
# requires v0.3.2+ — see CHANGELOG)
|
|
470
480
|
|
|
471
481
|
compression:
|
|
472
482
|
enabled: true
|
|
@@ -507,6 +517,9 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
|
|
|
507
517
|
| `/api/decision/invalidate` | POST | Mark decision as invalid |
|
|
508
518
|
| `/api/graph/{id}` | GET | Session graph stats |
|
|
509
519
|
| `/api/graph/{id}/html` | GET | **Interactive graph page** — decision-history timeline, supersession arcs, type/status filters, search, zoom/pan, PNG export. Zero external dependencies (works offline) |
|
|
520
|
+
| `/api/graph/{id}/why?q=` | GET | **Reasoning:** causal chain behind a decision (old → new with trigger/reason/evidence) |
|
|
521
|
+
| `/api/graph/{id}/reasoning` | GET | **Reasoning view:** active decisions by topic, recent changes, consistency audit |
|
|
522
|
+
| `/api/ontology` | GET | Machine-readable graph ontology (types, relations, status state machine) |
|
|
510
523
|
| `/api/stats` | GET | Token savings analytics |
|
|
511
524
|
| `/health` | GET | Health check |
|
|
512
525
|
| `/docs` | GET | Swagger UI |
|
|
@@ -517,15 +530,13 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
|
|
|
517
530
|
|
|
518
531
|
- API key auth — `TOKENMIZER_API_KEY` (constant-time comparison)
|
|
519
532
|
- Secret/PII redaction applied once at ingestion, before graph storage,
|
|
520
|
-
checkpoint storage,
|
|
521
|
-
extraction model
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
The checkpoint layer independently re-redacts what it persists
|
|
528
|
-
(defense-in-depth, added in the same audit).
|
|
533
|
+
checkpoint storage, and every LLM call (main chat and the background
|
|
534
|
+
extraction model). Patterns cover Anthropic/OpenAI/Google/GitHub/AWS/
|
|
535
|
+
Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded credentials
|
|
536
|
+
(`postgres://user:pass@host`), and generic `key=`/`password=`
|
|
537
|
+
assignments. Best-effort by nature — an unrecognized format with no
|
|
538
|
+
keyword context can still slip through. The checkpoint layer
|
|
539
|
+
independently re-redacts what it persists (defense in depth).
|
|
529
540
|
- Session-isolated cache (sensitive data never shared across sessions)
|
|
530
541
|
- Basic prompt-injection keyword filter — catches copy-pasted jailbreak
|
|
531
542
|
templates only; **not** a security boundary against a motivated
|
|
@@ -78,13 +78,22 @@ Your App → TokenMizer (:8000) → Claude / GPT / Gemini / any LLM
|
|
|
78
78
|
| 🔴 `INVALIDATED` | Explicitly wrong/cancelled | ⚠️ Always (warning) |
|
|
79
79
|
| ⬜ `ARCHIVED` | Superseded >7 days ago — aged out | ❌ Never |
|
|
80
80
|
|
|
81
|
-
History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable
|
|
81
|
+
History is **never deleted**. "Why did we switch from React to Next.js?" — always answerable:
|
|
82
|
+
ask `GET /api/graph/{session}/why?q=react` (or the `why_decision` MCP tool) and get the full
|
|
83
|
+
old → new trail with trigger, reason, and evidence per hop.
|
|
82
84
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
85
|
+
### From Storage to Reasoning
|
|
86
|
+
|
|
87
|
+
The graph doesn't just store facts — it answers questions over them:
|
|
88
|
+
|
|
89
|
+
| Capability | Endpoint / Tool | What it answers |
|
|
90
|
+
|---|---|---|
|
|
91
|
+
| **Ontology** | `GET /api/ontology` | The formal vocabulary: node/edge types with semantics, and the status state machine (which lifecycle transitions are legal) |
|
|
92
|
+
| **Causal chains** | `GET /api/graph/{id}/why?q=...` · MCP `why_decision` | "Why is X the current choice?" — walks the supersession chain with trigger/reason/evidence per hop |
|
|
93
|
+
| **Reasoning view** | `GET /api/graph/{id}/reasoning` | Active decisions per topic, recent changes, decision timeline, and a consistency audit |
|
|
94
|
+
| **Consistency audit** | (part of `/reasoning`) | Contradictions the tracker missed, superseded decisions with lost history, dangling references |
|
|
95
|
+
|
|
96
|
+
All reasoning is deterministic and local — no LLM calls, no extra cost.
|
|
88
97
|
|
|
89
98
|
---
|
|
90
99
|
|
|
@@ -285,10 +294,15 @@ env = { TOKENMIZER_URL = "http://localhost:8000" }
|
|
|
285
294
|
</details>
|
|
286
295
|
|
|
287
296
|
Then restart the client. Keep `tokenmizer serve` running for the
|
|
288
|
-
checkpoint/resume/stats tools (file analysis works without it).
|
|
297
|
+
checkpoint/resume/stats/reasoning tools (file analysis works without it).
|
|
289
298
|
If `tokenmizer-mcp` isn't on your PATH, use `"command": "python"`,
|
|
290
299
|
`"args": ["-m", "tokenmizer.mcp.server"]` instead.
|
|
291
300
|
|
|
301
|
+
**Tools exposed (6):** `checkpoint_session`, `resume_session`,
|
|
302
|
+
`get_graph_stats`, `analyze_file`, `get_savings_stats`, and
|
|
303
|
+
`why_decision` — ask your agent *"why did we pick X?"* and it traces the
|
|
304
|
+
decision's supersession chain with reasons and evidence.
|
|
305
|
+
|
|
292
306
|
---
|
|
293
307
|
|
|
294
308
|
## Other Tools
|
|
@@ -386,12 +400,8 @@ graph_checkpoint:
|
|
|
386
400
|
enabled: true
|
|
387
401
|
trigger_at_percent: 0.85
|
|
388
402
|
use_llm_extraction: false # true = hybrid LLM+heuristic extraction
|
|
389
|
-
# (needs a provider key, ~$0.001/turn
|
|
390
|
-
#
|
|
391
|
-
# nothing — the call site passed provider_fn
|
|
392
|
-
# to the wrong function and raised TypeError
|
|
393
|
-
# on every turn, falling back to heuristics.
|
|
394
|
-
# Fixed + regression-tested; see CHANGELOG.
|
|
403
|
+
# (needs a provider key, ~$0.001/turn;
|
|
404
|
+
# requires v0.3.2+ — see CHANGELOG)
|
|
395
405
|
|
|
396
406
|
compression:
|
|
397
407
|
enabled: true
|
|
@@ -432,6 +442,9 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
|
|
|
432
442
|
| `/api/decision/invalidate` | POST | Mark decision as invalid |
|
|
433
443
|
| `/api/graph/{id}` | GET | Session graph stats |
|
|
434
444
|
| `/api/graph/{id}/html` | GET | **Interactive graph page** — decision-history timeline, supersession arcs, type/status filters, search, zoom/pan, PNG export. Zero external dependencies (works offline) |
|
|
445
|
+
| `/api/graph/{id}/why?q=` | GET | **Reasoning:** causal chain behind a decision (old → new with trigger/reason/evidence) |
|
|
446
|
+
| `/api/graph/{id}/reasoning` | GET | **Reasoning view:** active decisions by topic, recent changes, consistency audit |
|
|
447
|
+
| `/api/ontology` | GET | Machine-readable graph ontology (types, relations, status state machine) |
|
|
435
448
|
| `/api/stats` | GET | Token savings analytics |
|
|
436
449
|
| `/health` | GET | Health check |
|
|
437
450
|
| `/docs` | GET | Swagger UI |
|
|
@@ -442,15 +455,13 @@ TOKENMIZER_API_KEY=strong-key docker-compose up
|
|
|
442
455
|
|
|
443
456
|
- API key auth — `TOKENMIZER_API_KEY` (constant-time comparison)
|
|
444
457
|
- Secret/PII redaction applied once at ingestion, before graph storage,
|
|
445
|
-
checkpoint storage,
|
|
446
|
-
extraction model
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
The checkpoint layer independently re-redacts what it persists
|
|
453
|
-
(defense-in-depth, added in the same audit).
|
|
458
|
+
checkpoint storage, and every LLM call (main chat and the background
|
|
459
|
+
extraction model). Patterns cover Anthropic/OpenAI/Google/GitHub/AWS/
|
|
460
|
+
Slack/Stripe/JWT/OpenRouter/HF/xAI keys, URL-embedded credentials
|
|
461
|
+
(`postgres://user:pass@host`), and generic `key=`/`password=`
|
|
462
|
+
assignments. Best-effort by nature — an unrecognized format with no
|
|
463
|
+
keyword context can still slip through. The checkpoint layer
|
|
464
|
+
independently re-redacts what it persists (defense in depth).
|
|
454
465
|
- Session-isolated cache (sensitive data never shared across sessions)
|
|
455
466
|
- Basic prompt-injection keyword filter — catches copy-pasted jailbreak
|
|
456
467
|
templates only; **not** a security boundary against a motivated
|
|
@@ -108,8 +108,8 @@ def main() -> int:
|
|
|
108
108
|
r = rpc("tools/list", req_id=2)
|
|
109
109
|
tools = {t["name"] for t in r.get("result", {}).get("tools", [])}
|
|
110
110
|
expected = {"checkpoint_session", "resume_session", "get_graph_stats",
|
|
111
|
-
"analyze_file", "get_savings_stats"}
|
|
112
|
-
check("tools/list exposes all
|
|
111
|
+
"analyze_file", "get_savings_stats", "why_decision"}
|
|
112
|
+
check("tools/list exposes all 6 tools", tools == expected, str(tools))
|
|
113
113
|
|
|
114
114
|
# 3. Local tool (no proxy): analyze_file on a real CSV
|
|
115
115
|
with tempfile.NamedTemporaryFile("w", suffix=".csv", delete=False,
|
|
@@ -137,8 +137,21 @@ def main() -> int:
|
|
|
137
137
|
text = r.get("result", {}).get("content", [{}])[0].get("text", "")
|
|
138
138
|
check("resume_session returns context", "TokenMizer Resume" in text, text[:160])
|
|
139
139
|
|
|
140
|
+
# Reasoning tool round-trips through the proxy. Whether the test
|
|
141
|
+
# session happens to contain a matching decision or not, a healthy
|
|
142
|
+
# server answers with a trail or a clean "no match" — never an error.
|
|
143
|
+
r = rpc("tools/call", {"name": "why_decision",
|
|
144
|
+
"arguments": {"session_id": "mcp-e2e-test",
|
|
145
|
+
"query": "postgres"}}, req_id=7)
|
|
146
|
+
res = r.get("result", {})
|
|
147
|
+
text = res.get("content", [{}])[0].get("text", "")
|
|
148
|
+
check("why_decision answers without error",
|
|
149
|
+
res.get("isError") is False
|
|
150
|
+
and ("Decision trail" in text or "No decision matching" in text),
|
|
151
|
+
text[:160])
|
|
152
|
+
|
|
140
153
|
# 5. Unknown method → JSON-RPC error, not crash
|
|
141
|
-
r = rpc("bogus/method", {}, req_id=
|
|
154
|
+
r = rpc("bogus/method", {}, req_id=8)
|
|
142
155
|
check("unknown method returns -32601 error",
|
|
143
156
|
r.get("error", {}).get("code") == -32601)
|
|
144
157
|
|
|
@@ -6,12 +6,12 @@
|
|
|
6
6
|
"url": "https://github.com/Shweta-Mishra-ai/tokenmizer",
|
|
7
7
|
"source": "github"
|
|
8
8
|
},
|
|
9
|
-
"version": "0.
|
|
9
|
+
"version": "0.4.0",
|
|
10
10
|
"packages": [
|
|
11
11
|
{
|
|
12
12
|
"registryType": "pypi",
|
|
13
13
|
"identifier": "tokenmizer",
|
|
14
|
-
"version": "0.
|
|
14
|
+
"version": "0.4.0",
|
|
15
15
|
"transport": { "type": "stdio" },
|
|
16
16
|
"environmentVariables": [
|
|
17
17
|
{
|
|
@@ -194,10 +194,8 @@ class TestHealthAndDocs:
|
|
|
194
194
|
def test_graph_share_html(self, client):
|
|
195
195
|
"""Shareable graph page: self-contained interactive HTML.
|
|
196
196
|
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
history (transitions), which the old template computed and then
|
|
200
|
-
silently never used.
|
|
197
|
+
The artifact must work offline (no CDN/external loads) and must
|
|
198
|
+
embed the supersession history (transitions) it renders.
|
|
201
199
|
"""
|
|
202
200
|
c, _ = client
|
|
203
201
|
# Put something in the graph via the pipeline first
|
|
@@ -223,11 +221,10 @@ class TestHealthAndDocs:
|
|
|
223
221
|
|
|
224
222
|
class TestInvalidateDecisionScope:
|
|
225
223
|
"""
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
are now eligible, and the response lists exactly what was affected.
|
|
224
|
+
/api/decision/invalidate must only match ACTIVE (COMPLETED) decisions:
|
|
225
|
+
matching across all statuses would overwrite SUPERSEDED history nodes
|
|
226
|
+
and destroy their supersession record. The response must list every
|
|
227
|
+
affected node.
|
|
231
228
|
"""
|
|
232
229
|
|
|
233
230
|
def _seed_graph(self, session_id, tmp_path):
|
|
@@ -1,14 +1,10 @@
|
|
|
1
1
|
"""
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
hybrid_extractor.py knew "supabase" but this file didn't)
|
|
9
|
-
- "Use Clerk for authentication" → None (same gap)
|
|
10
|
-
- "Use FastAPI with SQLAlchemy and PostgreSQL" → only "web_framework";
|
|
11
|
-
a later Postgres→SQLite switch was never detected as contradicting it.
|
|
2
|
+
Tests for the decision topic classifier and same-decision detection.
|
|
3
|
+
|
|
4
|
+
Covers the known-hard cases: the imperative "Go with X" vs. Go-the-language
|
|
5
|
+
ambiguity, technology names shared with the extractor vocabulary, multi-topic
|
|
6
|
+
statements, bigram/single-word precedence, and near-duplicate label
|
|
7
|
+
containment.
|
|
12
8
|
"""
|
|
13
9
|
from tokenmizer.graph_memory.decision_tracker import (
|
|
14
10
|
classify_topic,
|
|
@@ -112,12 +108,12 @@ def test_same_decision_not_superseded(tmp_path):
|
|
|
112
108
|
assert old_id not in hits
|
|
113
109
|
|
|
114
110
|
|
|
115
|
-
# ──
|
|
111
|
+
# ── Near-duplicate decision merging ──────────────────────────────────────────
|
|
116
112
|
|
|
117
113
|
def test_containment_variant_merges_not_supersedes(tmp_path):
|
|
118
|
-
"""
|
|
119
|
-
|
|
120
|
-
|
|
114
|
+
"""Two label variants of one decision (emitted from a single message)
|
|
115
|
+
must merge into one node rather than supersede each other, which would
|
|
116
|
+
record a spurious decision change."""
|
|
121
117
|
g = GraphMemory(session_id="t-dup", storage_dir=str(tmp_path))
|
|
122
118
|
id1 = g.add_node(NodeType.DECISION, "use React for the frontend.",
|
|
123
119
|
NodeStatus.COMPLETED)
|
|
@@ -203,11 +203,10 @@ class TestPersistence:
|
|
|
203
203
|
|
|
204
204
|
class TestArchivedReachability:
|
|
205
205
|
"""
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS days (from supersession time).
|
|
206
|
+
SUPERSEDED decisions age into ARCHIVED after
|
|
207
|
+
GraphMemory.ARCHIVE_SUPERSEDED_AFTER_DAYS, measured from supersession
|
|
208
|
+
time. apply_importance_decay() is the only path that sets ARCHIVED,
|
|
209
|
+
so these tests guard the state's reachability.
|
|
211
210
|
"""
|
|
212
211
|
|
|
213
212
|
def test_superseded_decision_ages_into_archived(self, graph):
|
|
@@ -151,15 +151,10 @@ async def test_extract_without_provider():
|
|
|
151
151
|
@pytest.mark.asyncio
|
|
152
152
|
async def test_extract_llm_pass_actually_invoked_app_style():
|
|
153
153
|
"""
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
was caught by the broad except, and logged as a provider failure.
|
|
159
|
-
|
|
160
|
-
This test replicates the app.py call pattern exactly as fixed:
|
|
161
|
-
construct with defaults, pass provider_fn to extract(), and assert
|
|
162
|
-
the provider was actually called and its output merged in.
|
|
154
|
+
Regression guard for the api/app.py call pattern: construct with
|
|
155
|
+
defaults and pass provider_fn to extract(). Asserts the provider is
|
|
156
|
+
actually invoked and its output is merged — a call-signature drift
|
|
157
|
+
here silently disables the LLM pass.
|
|
163
158
|
"""
|
|
164
159
|
calls = []
|
|
165
160
|
|
|
@@ -183,10 +178,9 @@ async def test_extract_llm_pass_actually_invoked_app_style():
|
|
|
183
178
|
|
|
184
179
|
class TestMinConfidenceFilter:
|
|
185
180
|
"""
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
only). Default 0.55 keeps everything (backward compatible).
|
|
181
|
+
min_confidence filters extract() output by merge()'s confidence tiers
|
|
182
|
+
(0.95 corroborated / 0.80 LLM-only / 0.65 heuristic-only). The default
|
|
183
|
+
of 0.55 keeps every tier.
|
|
190
184
|
"""
|
|
191
185
|
|
|
192
186
|
def test_default_keeps_heuristic_only_items(self):
|
|
@@ -1,16 +1,13 @@
|
|
|
1
1
|
"""
|
|
2
|
-
Regression tests for the MCP stdio server
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
1. isError
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
3.
|
|
11
|
-
the process with no JSON-RPC error — the client hung, then saw the
|
|
12
|
-
subprocess die.
|
|
13
|
-
4. Malformed JSON lines were silently dropped (no log, no error response).
|
|
2
|
+
Regression tests for the MCP stdio server.
|
|
3
|
+
|
|
4
|
+
Invariants under test:
|
|
5
|
+
1. isError is structural — validation failures and handler crashes are
|
|
6
|
+
reported with isError: true regardless of message text.
|
|
7
|
+
2. No input terminates the read loop: malformed JSON returns -32700,
|
|
8
|
+
non-object messages return -32600, handler exceptions return -32603,
|
|
9
|
+
and subsequent requests are still served.
|
|
10
|
+
3. Required arguments are validated with typed, descriptive errors.
|
|
14
11
|
"""
|
|
15
12
|
import io
|
|
16
13
|
import json
|
|
@@ -81,14 +78,14 @@ def test_analyze_file_success_not_error(tmp_path):
|
|
|
81
78
|
|
|
82
79
|
|
|
83
80
|
def test_handler_crash_is_error_not_exception(monkeypatch):
|
|
84
|
-
def
|
|
85
|
-
raise RuntimeError("
|
|
81
|
+
def failing_handler(args):
|
|
82
|
+
raise RuntimeError("simulated handler failure")
|
|
86
83
|
# handle_tool_call builds its dispatch dict from module globals at call
|
|
87
84
|
# time, so patching the module attribute is picked up.
|
|
88
|
-
monkeypatch.setattr(mcp, "handle_get_savings_stats",
|
|
85
|
+
monkeypatch.setattr(mcp, "handle_get_savings_stats", failing_handler)
|
|
89
86
|
text, is_error = mcp.handle_tool_call("get_savings_stats", {})
|
|
90
87
|
assert is_error is True
|
|
91
|
-
assert "
|
|
88
|
+
assert "simulated handler failure" in text or "internal error" in text.lower()
|
|
92
89
|
|
|
93
90
|
|
|
94
91
|
# ── stdio transport: survives hostile input ──────────────────────────────────
|
|
@@ -120,7 +117,7 @@ def test_stdio_survives_malformed_json(monkeypatch):
|
|
|
120
117
|
])
|
|
121
118
|
assert out[0]["error"]["code"] == -32700
|
|
122
119
|
assert out[1]["id"] == 2
|
|
123
|
-
assert len(out[1]["result"]["tools"]) ==
|
|
120
|
+
assert len(out[1]["result"]["tools"]) == 6
|
|
124
121
|
|
|
125
122
|
|
|
126
123
|
def test_stdio_survives_non_object_json(monkeypatch):
|
|
@@ -159,7 +156,7 @@ def test_stdio_handler_exception_yields_jsonrpc_error(monkeypatch):
|
|
|
159
156
|
|
|
160
157
|
|
|
161
158
|
def test_stdio_missing_arg_reports_is_error_true(monkeypatch):
|
|
162
|
-
"""
|
|
159
|
+
"""A missing required argument must reach the client as isError: true."""
|
|
163
160
|
out = _run_lines(monkeypatch, [
|
|
164
161
|
json.dumps({"jsonrpc": "2.0", "id": 6, "method": "tools/call",
|
|
165
162
|
"params": {"name": "checkpoint_session", "arguments": {}}}),
|