tokenhub 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokenhub-0.1.0/LICENSE +21 -0
- tokenhub-0.1.0/PKG-INFO +230 -0
- tokenhub-0.1.0/README.md +213 -0
- tokenhub-0.1.0/backend/tokenhub/__init__.py +1 -0
- tokenhub-0.1.0/backend/tokenhub/__main__.py +8 -0
- tokenhub-0.1.0/backend/tokenhub/analytics/__init__.py +1 -0
- tokenhub-0.1.0/backend/tokenhub/analytics/breakdown.py +78 -0
- tokenhub-0.1.0/backend/tokenhub/analytics/service.py +21 -0
- tokenhub-0.1.0/backend/tokenhub/api/__init__.py +7 -0
- tokenhub-0.1.0/backend/tokenhub/api/container.py +146 -0
- tokenhub-0.1.0/backend/tokenhub/api/routes.py +267 -0
- tokenhub-0.1.0/backend/tokenhub/app.py +77 -0
- tokenhub-0.1.0/backend/tokenhub/cli.py +65 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/__init__.py +1 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/__init__.py +3 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/connector.py +107 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/parser.py +238 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/claude/__init__.py +3 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/claude/connector.py +152 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/claude/parser.py +182 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/codex/__init__.py +3 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/codex/connector.py +209 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/codex/parser.py +261 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/copilot/__init__.py +3 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/copilot/connector.py +155 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/copilot/parser.py +224 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/hermes/__init__.py +3 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/hermes/connector.py +106 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/hermes/parser.py +246 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/metadata.py +11 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/protocol.py +116 -0
- tokenhub-0.1.0/backend/tokenhub/connectors/registry.py +42 -0
- tokenhub-0.1.0/backend/tokenhub/database/__init__.py +11 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/env.py +58 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0001_initial.py +85 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0002_cursor_prefix_fingerprint.py +29 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0003_source_trust_and_quality.py +44 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0004_auto_import_roots.py +25 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0005_usage_metadata.py +24 -0
- tokenhub-0.1.0/backend/tokenhub/database/migrations.py +47 -0
- tokenhub-0.1.0/backend/tokenhub/database/models.py +87 -0
- tokenhub-0.1.0/backend/tokenhub/database/repositories.py +491 -0
- tokenhub-0.1.0/backend/tokenhub/database/session.py +35 -0
- tokenhub-0.1.0/backend/tokenhub/discovery/__init__.py +1 -0
- tokenhub-0.1.0/backend/tokenhub/discovery/service.py +51 -0
- tokenhub-0.1.0/backend/tokenhub/domain/__init__.py +17 -0
- tokenhub-0.1.0/backend/tokenhub/domain/models.py +273 -0
- tokenhub-0.1.0/backend/tokenhub/ingestion/__init__.py +1 -0
- tokenhub-0.1.0/backend/tokenhub/ingestion/collection.py +116 -0
- tokenhub-0.1.0/backend/tokenhub/ingestion/service.py +137 -0
- tokenhub-0.1.0/backend/tokenhub/runtime/__init__.py +5 -0
- tokenhub-0.1.0/backend/tokenhub/runtime/client.py +69 -0
- tokenhub-0.1.0/backend/tokenhub/runtime/control.py +62 -0
- tokenhub-0.1.0/backend/tokenhub/runtime/instance.py +79 -0
- tokenhub-0.1.0/backend/tokenhub/runtime/manager.py +161 -0
- tokenhub-0.1.0/backend/tokenhub/security/__init__.py +6 -0
- tokenhub-0.1.0/backend/tokenhub/security/http.py +108 -0
- tokenhub-0.1.0/backend/tokenhub/security/paths.py +204 -0
- tokenhub-0.1.0/backend/tokenhub/security/redaction.py +13 -0
- tokenhub-0.1.0/backend/tokenhub/security/windows_paths.py +223 -0
- tokenhub-0.1.0/backend/tokenhub/server.py +29 -0
- tokenhub-0.1.0/backend/tokenhub/settings.py +53 -0
- tokenhub-0.1.0/backend/tokenhub/web/assets/index-B2a5VymX.css +1 -0
- tokenhub-0.1.0/backend/tokenhub/web/assets/index-Ctyi0WkJ.js +40 -0
- tokenhub-0.1.0/backend/tokenhub/web/favicon.svg +4 -0
- tokenhub-0.1.0/backend/tokenhub/web/index.html +16 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/PKG-INFO +230 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/SOURCES.txt +73 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/dependency_links.txt +1 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/entry_points.txt +2 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/requires.txt +7 -0
- tokenhub-0.1.0/backend/tokenhub.egg-info/top_level.txt +1 -0
- tokenhub-0.1.0/pyproject.toml +52 -0
- tokenhub-0.1.0/setup.cfg +4 -0
- tokenhub-0.1.0/tests/test_package.py +5 -0
tokenhub-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Aldrin Joseph
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
tokenhub-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tokenhub
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Local-first TokenHub
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Requires-Python: >=3.12
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Requires-Dist: fastapi
|
|
10
|
+
Requires-Dist: uvicorn[standard]
|
|
11
|
+
Requires-Dist: pydantic
|
|
12
|
+
Requires-Dist: sqlalchemy
|
|
13
|
+
Requires-Dist: alembic
|
|
14
|
+
Requires-Dist: filelock<4,>=3.12
|
|
15
|
+
Requires-Dist: psutil<8,>=5.9
|
|
16
|
+
Dynamic: license-file
|
|
17
|
+
|
|
18
|
+
# TokenHub
|
|
19
|
+
|
|
20
|
+
A local-first usage observatory. TokenHub detects the coding agents installed on
|
|
21
|
+
this machine, imports usage from sources you explicitly approve, and
|
|
22
|
+
shows agent, model, and session usage in a local explorer. Nothing leaves the machine: no provider
|
|
23
|
+
API is called, no provider executable is launched, and only usage metadata is retained. Credentials and provider accounts are never
|
|
24
|
+
read; transcript content is never stored or shown. Approved session files can contain chat text, which the parsers discard.
|
|
25
|
+
|
|
26
|
+
Supported sources: **OpenAI Codex** session usage, **Claude Code** project and
|
|
27
|
+
subagent session logs, **Hermes Agent** session counters in its local state
|
|
28
|
+
database, **VS Code Copilot Chat** saved sessions, and **Antigravity** conversation
|
|
29
|
+
databases.
|
|
30
|
+
|
|
31
|
+
## Install and run
|
|
32
|
+
|
|
33
|
+
Requires Python 3.12 or newer. After the public package is released:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install tokenhub
|
|
37
|
+
tokenhub # starts in the background and opens the local dashboard
|
|
38
|
+
tokenhub status # prints its local URL
|
|
39
|
+
tokenhub open # opens an already running dashboard
|
|
40
|
+
tokenhub stop # stops this TokenHub instance
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`tokenhub start --no-open` starts without opening a browser. Use
|
|
44
|
+
`tokenhub start --port 9000` to choose another loopback port. A second start
|
|
45
|
+
reuses the running instance, including its chosen port. The server continues
|
|
46
|
+
after the terminal closes; run `tokenhub` again after a reboot or login. Logs
|
|
47
|
+
and local data are stored in `~/.tokenhub/`, with startup details in
|
|
48
|
+
`~/.tokenhub/runtime.log`.
|
|
49
|
+
|
|
50
|
+
For development from this checkout, run `uv sync --all-groups` and
|
|
51
|
+
`uv run tokenhub --no-open`.
|
|
52
|
+
|
|
53
|
+
TokenHub binds to `127.0.0.1:7432` by default and prints that URL when it
|
|
54
|
+
starts. It is loopback-only by design:
|
|
55
|
+
|
|
56
|
+
- the bind host is validated at startup — `0.0.0.0`, `::`, or any non-loopback
|
|
57
|
+
value raises rather than serving;
|
|
58
|
+
- every request must carry a loopback `Host` header (`localhost`, `127.0.0.1`,
|
|
59
|
+
`[::1]`, with the configured port), so a DNS-rebinding page cannot reach the
|
|
60
|
+
API even from the local browser;
|
|
61
|
+
- state-changing requests (`approve`, `rescan`, `rebuild`) must also carry the
|
|
62
|
+
request's own origin; otherwise they are refused with `403`;
|
|
63
|
+
- no CORS headers are ever added, and `X-Forwarded-*` headers are ignored.
|
|
64
|
+
|
|
65
|
+
The UI is bundled with the Python package. To refresh it while developing:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
npm --prefix frontend ci
|
|
69
|
+
npm --prefix frontend run build # writes backend/tokenhub/web
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The packaged UI is mounted at `/`; the API also works if the built UI is absent.
|
|
73
|
+
|
|
74
|
+
## Automatic collection
|
|
75
|
+
|
|
76
|
+
While TokenHub is running, it scans approved sources from all five providers at startup and every
|
|
77
|
+
30 seconds. The explorer refreshes itself every 10 seconds. Approving a provider
|
|
78
|
+
in the UI also starts its first import immediately. Unchanged session files are skipped. Hermes and Antigravity databases are checked on each scan
|
|
79
|
+
because active usage can live in their SQLite journal. Repeated scans do not
|
|
80
|
+
increase totals. Automatic scans of unchanged data do not create extra
|
|
81
|
+
import-history rows, including after restarting TokenHub.
|
|
82
|
+
|
|
83
|
+
Choose **Approve and import** on a provider card to approve its discovered
|
|
84
|
+
usage folder once and include both existing and future sessions automatically.
|
|
85
|
+
This consent survives restarts and is pinned to the folder's device and inode;
|
|
86
|
+
replacing the folder or redirecting it through a symlink cannot authorize another
|
|
87
|
+
location. **Stop including new sessions** removes that folder consent; previously
|
|
88
|
+
approved files keep updating. Without folder consent, new files require approval.
|
|
89
|
+
|
|
90
|
+
`GET /api/v1/collection` reports the scan interval, folder-consent state, last
|
|
91
|
+
completed scan, and failed-source count. Same-origin `POST` requests to
|
|
92
|
+
`/api/v1/collection/{provider}/enable` and `/disable` change consent (`codex`,
|
|
93
|
+
`claude_code`, `hermes`, `vscode_copilot`, or `antigravity`). Responses contain
|
|
94
|
+
no private paths. Failed sources are retried without preventing other sources
|
|
95
|
+
from updating, and collection runs even when the dashboard is closed.
|
|
96
|
+
|
|
97
|
+
For Codex, only per-response `token_usage_record.payload.usage` values contribute
|
|
98
|
+
to totals.
|
|
99
|
+
Known session messages are skipped. Cumulative token snapshots, malformed records,
|
|
100
|
+
and unknown usage structures remain visible as unsupported records; cumulative
|
|
101
|
+
snapshots are excluded from totals to avoid double counting. Workload is
|
|
102
|
+
input plus output, with cached input and reasoning already included in those
|
|
103
|
+
counts. Token counts are not billing amounts. The UI uses K/M/B abbreviations
|
|
104
|
+
and shows the exact count on hover.
|
|
105
|
+
|
|
106
|
+
Claude Code reads `message.usage` from assistant records under its `projects`
|
|
107
|
+
folder. Repeated streaming chunks are consolidated by message ID, copied history in
|
|
108
|
+
resumed/forked session files is counted once, and later usage updates replace
|
|
109
|
+
the session's previous normalized records. Hermes reads
|
|
110
|
+
only allowlisted session identifiers, timestamps, and token counters from a
|
|
111
|
+
no-follow snapshot of `state.db` and its journal; no provider database is written.
|
|
112
|
+
Each Hermes session contributes once, with running totals replaced atomically as
|
|
113
|
+
they change. Cache reads and writes are included in the input total for Claude
|
|
114
|
+
and Hermes because both store them in separate buckets. Hermes session
|
|
115
|
+
contributions are marked `high` quality; Claude message usage is marked `exact`.
|
|
116
|
+
|
|
117
|
+
VS Code Copilot Chat imports saved `chatSessions/*.jsonl` requests from VS Code's
|
|
118
|
+
workspace storage. It records each Copilot request's model, session, input tokens,
|
|
119
|
+
and output tokens when VS Code saved those counters. Later session patches replace
|
|
120
|
+
earlier counters; unchanged files do not add usage again. Requests without token
|
|
121
|
+
counters remain unsupported rather than receiving estimated values. Cache and
|
|
122
|
+
reasoning counters are not available in these saved requests. Inline suggestions,
|
|
123
|
+
Copilot CLI, and Copilot activity in other editors are outside this source's scope.
|
|
124
|
+
`VSCODE_USER_DATA_DIR` can point to a different VS Code user-data directory,
|
|
125
|
+
such as an Insiders installation.
|
|
126
|
+
|
|
127
|
+
Antigravity imports per-generation model, timestamp, input, cache-read, output,
|
|
128
|
+
and reasoning counters from approved `~/.gemini/antigravity/conversations/*.db`
|
|
129
|
+
files. It reads a snapshot of the database and its active journal without writing
|
|
130
|
+
to the provider database. Antigravity's local metadata format is private, so
|
|
131
|
+
unrecognized or inconsistent generations are excluded and reported as partial
|
|
132
|
+
instead of estimated. `ANTIGRAVITY_HOME` can point to another Antigravity root.
|
|
133
|
+
|
|
134
|
+
## Discovery and approval
|
|
135
|
+
|
|
136
|
+
`GET /api/v1/discovery` reports each provider's display name, connection state,
|
|
137
|
+
a confidence level derived from its evidence (`high` when two or more
|
|
138
|
+
independent signals were found, `medium` for one, `low` for none), the evidence
|
|
139
|
+
codes it found (`executable_on_path`, `known_root_exists`, `configuration_found`,
|
|
140
|
+
`session_source_found`, `state_database_found`), and a stable path fingerprint.
|
|
141
|
+
Discovery reads **presence only** — it stats candidate roots and directories and
|
|
142
|
+
never opens a provider file.
|
|
143
|
+
|
|
144
|
+
The provider card combines approval and first import in one click. It also
|
|
145
|
+
enables future session imports for that provider. The lower-level API remains
|
|
146
|
+
available for one-source troubleshooting:
|
|
147
|
+
|
|
148
|
+
1. `POST /api/v1/sources/{source_id}/approve` persists the canonical path for
|
|
149
|
+
that one source. Before approval, responses contain no absolute path at all.
|
|
150
|
+
2. `POST /api/v1/sources/{source_id}/rescan` imports new records through the
|
|
151
|
+
connector's parser. Repeating a rescan over unchanged bytes inserts nothing;
|
|
152
|
+
`POST /api/v1/rebuild` re-derives all normalized events from scratch and
|
|
153
|
+
leaves totals unchanged.
|
|
154
|
+
|
|
155
|
+
`GET /api/v1/dashboard` and `GET /api/v1/data-quality` report observed usage.
|
|
156
|
+
Workload is input total plus output total; cache and reasoning are breakdowns,
|
|
157
|
+
never extra usage. A metric the data cannot support is returned as `null` and
|
|
158
|
+
rendered as an em dash — TokenHub never substitutes `0` for "unknown".
|
|
159
|
+
|
|
160
|
+
The default **Usage explorer** combines the consolidated token summary with
|
|
161
|
+
agent, model, and session breakdowns for Codex, Claude Code, Hermes Agent,
|
|
162
|
+
VS Code Copilot, and Antigravity. Filter by all time, today, yesterday, the last
|
|
163
|
+
7 or 30 days, or a selected local calendar date.
|
|
164
|
+
Choose an agent to rank its models by total, input, output, cache-read, or
|
|
165
|
+
reasoning tokens. Search for a model, select it to see its sessions, and expand
|
|
166
|
+
a session to compare the models used inside it. Exact counts are available on
|
|
167
|
+
hover. Source health and import quality appear together under **Local sources**.
|
|
168
|
+
|
|
169
|
+
`GET /api/v1/usage` returns canonical totals plus agent, model, and session
|
|
170
|
+
breakdowns. Session identifiers are opaque, stable hashes; original session IDs
|
|
171
|
+
and source paths are not exposed. The same Claude message deduplication is used
|
|
172
|
+
throughout the explorer. Optional timezone-aware `from` and `to` query bounds
|
|
173
|
+
filter usage by recorded event time; the upper bound is exclusive. Known older
|
|
174
|
+
parsers re-read approved sources on startup to add
|
|
175
|
+
model/session metadata without changing previously imported Codex counters.
|
|
176
|
+
|
|
177
|
+
## Data boundaries
|
|
178
|
+
|
|
179
|
+
- Only usage metadata is normalized. Prompts, messages, tool output, transcript
|
|
180
|
+
bodies, and raw provider records are never retained or returned by the API.
|
|
181
|
+
- Provider credentials, API keys, `auth.json`, keychain, and cookie stores are
|
|
182
|
+
never read. Approved session JSONL can contain chat text; only allowlisted
|
|
183
|
+
usage fields are normalized, and raw content is never stored or returned.
|
|
184
|
+
- Anything outside an approved source's own approved root: canonical-path
|
|
185
|
+
validation rejects symlink escapes, and the approved root is re-validated by
|
|
186
|
+
device/inode at every scan.
|
|
187
|
+
|
|
188
|
+
## Tests
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
uv run pytest -q # unit + API
|
|
192
|
+
uv run pytest tests/integration -m integration -q # real app stack, synthetic homes
|
|
193
|
+
uv run ruff check backend tests
|
|
194
|
+
uv run mypy backend
|
|
195
|
+
cd frontend && npm run test -- --run && npm run build
|
|
196
|
+
uv build
|
|
197
|
+
python scripts/verify_distribution.py dist/* # tests installed wheel and source archive
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
Every test is offline and deterministic: synthetic fixtures in a temp directory,
|
|
201
|
+
no real home directory, no network, no provider subprocess. A test that points
|
|
202
|
+
at a real provider path or opens a real credential store is a bug.
|
|
203
|
+
|
|
204
|
+
## Layout
|
|
205
|
+
|
|
206
|
+
| Path | Responsibility |
|
|
207
|
+
|---|---|
|
|
208
|
+
| `backend/tokenhub/domain/` | provider-neutral enums and data models |
|
|
209
|
+
| `backend/tokenhub/security/` | path containment, redaction, Host/Origin policy |
|
|
210
|
+
| `backend/tokenhub/connectors/` | one module per provider behind a shared contract |
|
|
211
|
+
| `backend/tokenhub/discovery/` | presence discovery and the pre-approval handoff |
|
|
212
|
+
| `backend/tokenhub/ingestion/` | approval gate, incremental scan, rebuild |
|
|
213
|
+
| `backend/tokenhub/database/` | SQLAlchemy models, repositories, Alembic migrations |
|
|
214
|
+
| `backend/tokenhub/analytics/` | dashboard aggregates over observed deltas |
|
|
215
|
+
| `backend/tokenhub/api/` | versioned `/api/v1` routes and the app container |
|
|
216
|
+
| `frontend/` | React + TypeScript + Vite client (Vitest, MSW) |
|
|
217
|
+
|
|
218
|
+
## Current limitations
|
|
219
|
+
|
|
220
|
+
- **Local schemas only.** Unknown usage structures and malformed records are
|
|
221
|
+
reported as partial or unavailable. Hermes needs a `sessions` table containing
|
|
222
|
+
session IDs, start timestamps, and input/output counters. An unsupported or
|
|
223
|
+
unreadable database leaves previously imported usage intact.
|
|
224
|
+
- **No cost or pricing.** TokenHub reports token counts only — no currency, no
|
|
225
|
+
provider plan data.
|
|
226
|
+
- **Model attribution depends on recorded metadata.** Missing model names appear
|
|
227
|
+
in “Model not recorded”. Hermes exposes session totals with a reported model;
|
|
228
|
+
model switches within that session cannot be split by its current counters.
|
|
229
|
+
- **No trend chart.** Date filters use observed usage timestamps. TokenHub does
|
|
230
|
+
not guess daily allocations for counters that lack event dates.
|
tokenhub-0.1.0/README.md
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
# TokenHub
|
|
2
|
+
|
|
3
|
+
A local-first usage observatory. TokenHub detects the coding agents installed on
|
|
4
|
+
this machine, imports usage from sources you explicitly approve, and
|
|
5
|
+
shows agent, model, and session usage in a local explorer. Nothing leaves the machine: no provider
|
|
6
|
+
API is called, no provider executable is launched, and only usage metadata is retained. Credentials and provider accounts are never
|
|
7
|
+
read; transcript content is never stored or shown. Approved session files can contain chat text, which the parsers discard.
|
|
8
|
+
|
|
9
|
+
Supported sources: **OpenAI Codex** session usage, **Claude Code** project and
|
|
10
|
+
subagent session logs, **Hermes Agent** session counters in its local state
|
|
11
|
+
database, **VS Code Copilot Chat** saved sessions, and **Antigravity** conversation
|
|
12
|
+
databases.
|
|
13
|
+
|
|
14
|
+
## Install and run
|
|
15
|
+
|
|
16
|
+
Requires Python 3.12 or newer. After the public package is released:
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
pip install tokenhub
|
|
20
|
+
tokenhub # starts in the background and opens the local dashboard
|
|
21
|
+
tokenhub status # prints its local URL
|
|
22
|
+
tokenhub open # opens an already running dashboard
|
|
23
|
+
tokenhub stop # stops this TokenHub instance
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
`tokenhub start --no-open` starts without opening a browser. Use
|
|
27
|
+
`tokenhub start --port 9000` to choose another loopback port. A second start
|
|
28
|
+
reuses the running instance, including its chosen port. The server continues
|
|
29
|
+
after the terminal closes; run `tokenhub` again after a reboot or login. Logs
|
|
30
|
+
and local data are stored in `~/.tokenhub/`, with startup details in
|
|
31
|
+
`~/.tokenhub/runtime.log`.
|
|
32
|
+
|
|
33
|
+
For development from this checkout, run `uv sync --all-groups` and
|
|
34
|
+
`uv run tokenhub --no-open`.
|
|
35
|
+
|
|
36
|
+
TokenHub binds to `127.0.0.1:7432` by default and prints that URL when it
|
|
37
|
+
starts. It is loopback-only by design:
|
|
38
|
+
|
|
39
|
+
- the bind host is validated at startup — `0.0.0.0`, `::`, or any non-loopback
|
|
40
|
+
value raises rather than serving;
|
|
41
|
+
- every request must carry a loopback `Host` header (`localhost`, `127.0.0.1`,
|
|
42
|
+
`[::1]`, with the configured port), so a DNS-rebinding page cannot reach the
|
|
43
|
+
API even from the local browser;
|
|
44
|
+
- state-changing requests (`approve`, `rescan`, `rebuild`) must also carry the
|
|
45
|
+
request's own origin; otherwise they are refused with `403`;
|
|
46
|
+
- no CORS headers are ever added, and `X-Forwarded-*` headers are ignored.
|
|
47
|
+
|
|
48
|
+
The UI is bundled with the Python package. To refresh it while developing:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
npm --prefix frontend ci
|
|
52
|
+
npm --prefix frontend run build # writes backend/tokenhub/web
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The packaged UI is mounted at `/`; the API also works if the built UI is absent.
|
|
56
|
+
|
|
57
|
+
## Automatic collection
|
|
58
|
+
|
|
59
|
+
While TokenHub is running, it scans approved sources from all five providers at startup and every
|
|
60
|
+
30 seconds. The explorer refreshes itself every 10 seconds. Approving a provider
|
|
61
|
+
in the UI also starts its first import immediately. Unchanged session files are skipped. Hermes and Antigravity databases are checked on each scan
|
|
62
|
+
because active usage can live in their SQLite journal. Repeated scans do not
|
|
63
|
+
increase totals. Automatic scans of unchanged data do not create extra
|
|
64
|
+
import-history rows, including after restarting TokenHub.
|
|
65
|
+
|
|
66
|
+
Choose **Approve and import** on a provider card to approve its discovered
|
|
67
|
+
usage folder once and include both existing and future sessions automatically.
|
|
68
|
+
This consent survives restarts and is pinned to the folder's device and inode;
|
|
69
|
+
replacing the folder or redirecting it through a symlink cannot authorize another
|
|
70
|
+
location. **Stop including new sessions** removes that folder consent; previously
|
|
71
|
+
approved files keep updating. Without folder consent, new files require approval.
|
|
72
|
+
|
|
73
|
+
`GET /api/v1/collection` reports the scan interval, folder-consent state, last
|
|
74
|
+
completed scan, and failed-source count. Same-origin `POST` requests to
|
|
75
|
+
`/api/v1/collection/{provider}/enable` and `/disable` change consent (`codex`,
|
|
76
|
+
`claude_code`, `hermes`, `vscode_copilot`, or `antigravity`). Responses contain
|
|
77
|
+
no private paths. Failed sources are retried without preventing other sources
|
|
78
|
+
from updating, and collection runs even when the dashboard is closed.
|
|
79
|
+
|
|
80
|
+
For Codex, only per-response `token_usage_record.payload.usage` values contribute
|
|
81
|
+
to totals.
|
|
82
|
+
Known session messages are skipped. Cumulative token snapshots, malformed records,
|
|
83
|
+
and unknown usage structures remain visible as unsupported records; cumulative
|
|
84
|
+
snapshots are excluded from totals to avoid double counting. Workload is
|
|
85
|
+
input plus output, with cached input and reasoning already included in those
|
|
86
|
+
counts. Token counts are not billing amounts. The UI uses K/M/B abbreviations
|
|
87
|
+
and shows the exact count on hover.
|
|
88
|
+
|
|
89
|
+
Claude Code reads `message.usage` from assistant records under its `projects`
|
|
90
|
+
folder. Repeated streaming chunks are consolidated by message ID, copied history in
|
|
91
|
+
resumed/forked session files is counted once, and later usage updates replace
|
|
92
|
+
the session's previous normalized records. Hermes reads
|
|
93
|
+
only allowlisted session identifiers, timestamps, and token counters from a
|
|
94
|
+
no-follow snapshot of `state.db` and its journal; no provider database is written.
|
|
95
|
+
Each Hermes session contributes once, with running totals replaced atomically as
|
|
96
|
+
they change. Cache reads and writes are included in the input total for Claude
|
|
97
|
+
and Hermes because both store them in separate buckets. Hermes session
|
|
98
|
+
contributions are marked `high` quality; Claude message usage is marked `exact`.
|
|
99
|
+
|
|
100
|
+
VS Code Copilot Chat imports saved `chatSessions/*.jsonl` requests from VS Code's
|
|
101
|
+
workspace storage. It records each Copilot request's model, session, input tokens,
|
|
102
|
+
and output tokens when VS Code saved those counters. Later session patches replace
|
|
103
|
+
earlier counters; unchanged files do not add usage again. Requests without token
|
|
104
|
+
counters remain unsupported rather than receiving estimated values. Cache and
|
|
105
|
+
reasoning counters are not available in these saved requests. Inline suggestions,
|
|
106
|
+
Copilot CLI, and Copilot activity in other editors are outside this source's scope.
|
|
107
|
+
`VSCODE_USER_DATA_DIR` can point to a different VS Code user-data directory,
|
|
108
|
+
such as an Insiders installation.
|
|
109
|
+
|
|
110
|
+
Antigravity imports per-generation model, timestamp, input, cache-read, output,
|
|
111
|
+
and reasoning counters from approved `~/.gemini/antigravity/conversations/*.db`
|
|
112
|
+
files. It reads a snapshot of the database and its active journal without writing
|
|
113
|
+
to the provider database. Antigravity's local metadata format is private, so
|
|
114
|
+
unrecognized or inconsistent generations are excluded and reported as partial
|
|
115
|
+
instead of estimated. `ANTIGRAVITY_HOME` can point to another Antigravity root.
|
|
116
|
+
|
|
117
|
+
## Discovery and approval
|
|
118
|
+
|
|
119
|
+
`GET /api/v1/discovery` reports each provider's display name, connection state,
|
|
120
|
+
a confidence level derived from its evidence (`high` when two or more
|
|
121
|
+
independent signals were found, `medium` for one, `low` for none), the evidence
|
|
122
|
+
codes it found (`executable_on_path`, `known_root_exists`, `configuration_found`,
|
|
123
|
+
`session_source_found`, `state_database_found`), and a stable path fingerprint.
|
|
124
|
+
Discovery reads **presence only** — it stats candidate roots and directories and
|
|
125
|
+
never opens a provider file.
|
|
126
|
+
|
|
127
|
+
The provider card combines approval and first import in one click. It also
|
|
128
|
+
enables future session imports for that provider. The lower-level API remains
|
|
129
|
+
available for one-source troubleshooting:
|
|
130
|
+
|
|
131
|
+
1. `POST /api/v1/sources/{source_id}/approve` persists the canonical path for
|
|
132
|
+
that one source. Before approval, responses contain no absolute path at all.
|
|
133
|
+
2. `POST /api/v1/sources/{source_id}/rescan` imports new records through the
|
|
134
|
+
connector's parser. Repeating a rescan over unchanged bytes inserts nothing;
|
|
135
|
+
`POST /api/v1/rebuild` re-derives all normalized events from scratch and
|
|
136
|
+
leaves totals unchanged.
|
|
137
|
+
|
|
138
|
+
`GET /api/v1/dashboard` and `GET /api/v1/data-quality` report observed usage.
|
|
139
|
+
Workload is input total plus output total; cache and reasoning are breakdowns,
|
|
140
|
+
never extra usage. A metric the data cannot support is returned as `null` and
|
|
141
|
+
rendered as an em dash — TokenHub never substitutes `0` for "unknown".
|
|
142
|
+
|
|
143
|
+
The default **Usage explorer** combines the consolidated token summary with
|
|
144
|
+
agent, model, and session breakdowns for Codex, Claude Code, Hermes Agent,
|
|
145
|
+
VS Code Copilot, and Antigravity. Filter by all time, today, yesterday, the last
|
|
146
|
+
7 or 30 days, or a selected local calendar date.
|
|
147
|
+
Choose an agent to rank its models by total, input, output, cache-read, or
|
|
148
|
+
reasoning tokens. Search for a model, select it to see its sessions, and expand
|
|
149
|
+
a session to compare the models used inside it. Exact counts are available on
|
|
150
|
+
hover. Source health and import quality appear together under **Local sources**.
|
|
151
|
+
|
|
152
|
+
`GET /api/v1/usage` returns canonical totals plus agent, model, and session
|
|
153
|
+
breakdowns. Session identifiers are opaque, stable hashes; original session IDs
|
|
154
|
+
and source paths are not exposed. The same Claude message deduplication is used
|
|
155
|
+
throughout the explorer. Optional timezone-aware `from` and `to` query bounds
|
|
156
|
+
filter usage by recorded event time; the upper bound is exclusive. Known older
|
|
157
|
+
parsers re-read approved sources on startup to add
|
|
158
|
+
model/session metadata without changing previously imported Codex counters.
|
|
159
|
+
|
|
160
|
+
## Data boundaries
|
|
161
|
+
|
|
162
|
+
- Only usage metadata is normalized. Prompts, messages, tool output, transcript
|
|
163
|
+
bodies, and raw provider records are never retained or returned by the API.
|
|
164
|
+
- Provider credentials, API keys, `auth.json`, keychain, and cookie stores are
|
|
165
|
+
never read. Approved session JSONL can contain chat text; only allowlisted
|
|
166
|
+
usage fields are normalized, and raw content is never stored or returned.
|
|
167
|
+
- Anything outside an approved source's own approved root: canonical-path
|
|
168
|
+
validation rejects symlink escapes, and the approved root is re-validated by
|
|
169
|
+
device/inode at every scan.
|
|
170
|
+
|
|
171
|
+
## Tests
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
uv run pytest -q # unit + API
|
|
175
|
+
uv run pytest tests/integration -m integration -q # real app stack, synthetic homes
|
|
176
|
+
uv run ruff check backend tests
|
|
177
|
+
uv run mypy backend
|
|
178
|
+
cd frontend && npm run test -- --run && npm run build
|
|
179
|
+
uv build
|
|
180
|
+
python scripts/verify_distribution.py dist/* # tests installed wheel and source archive
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
Every test is offline and deterministic: synthetic fixtures in a temp directory,
|
|
184
|
+
no real home directory, no network, no provider subprocess. A test that points
|
|
185
|
+
at a real provider path or opens a real credential store is a bug.
|
|
186
|
+
|
|
187
|
+
## Layout
|
|
188
|
+
|
|
189
|
+
| Path | Responsibility |
|
|
190
|
+
|---|---|
|
|
191
|
+
| `backend/tokenhub/domain/` | provider-neutral enums and data models |
|
|
192
|
+
| `backend/tokenhub/security/` | path containment, redaction, Host/Origin policy |
|
|
193
|
+
| `backend/tokenhub/connectors/` | one module per provider behind a shared contract |
|
|
194
|
+
| `backend/tokenhub/discovery/` | presence discovery and the pre-approval handoff |
|
|
195
|
+
| `backend/tokenhub/ingestion/` | approval gate, incremental scan, rebuild |
|
|
196
|
+
| `backend/tokenhub/database/` | SQLAlchemy models, repositories, Alembic migrations |
|
|
197
|
+
| `backend/tokenhub/analytics/` | dashboard aggregates over observed deltas |
|
|
198
|
+
| `backend/tokenhub/api/` | versioned `/api/v1` routes and the app container |
|
|
199
|
+
| `frontend/` | React + TypeScript + Vite client (Vitest, MSW) |
|
|
200
|
+
|
|
201
|
+
## Current limitations
|
|
202
|
+
|
|
203
|
+
- **Local schemas only.** Unknown usage structures and malformed records are
|
|
204
|
+
reported as partial or unavailable. Hermes needs a `sessions` table containing
|
|
205
|
+
session IDs, start timestamps, and input/output counters. An unsupported or
|
|
206
|
+
unreadable database leaves previously imported usage intact.
|
|
207
|
+
- **No cost or pricing.** TokenHub reports token counts only — no currency, no
|
|
208
|
+
provider plan data.
|
|
209
|
+
- **Model attribution depends on recorded metadata.** Missing model names appear
|
|
210
|
+
in “Model not recorded”. Hermes exposes session totals with a reported model;
|
|
211
|
+
model switches within that session cannot be split by its current counters.
|
|
212
|
+
- **No trend chart.** Date filters use observed usage timestamps. TokenHub does
|
|
213
|
+
not guess daily allocations for counters that lack event dates.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""TokenHub package."""
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Provider-neutral analytics over normalized local events."""
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Agent, model, and session views of the canonical observed workload."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from collections import defaultdict
|
|
5
|
+
from datetime import UTC
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from tokenhub.database.models import UsageEventRecord
|
|
9
|
+
|
|
10
|
+
_COUNTERS = (
|
|
11
|
+
"input_total_tokens", "output_total_tokens", "cache_read_tokens",
|
|
12
|
+
"cache_write_tokens", "reasoning_tokens",
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _session_key(event: UsageEventRecord) -> str:
|
|
17
|
+
# Stable across refreshes, with no original session id or filesystem path.
|
|
18
|
+
identity = f"{event.connector_id}\0{event.session_id or event.source_id}"
|
|
19
|
+
return hashlib.sha256(identity.encode()).hexdigest()[:24]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _summary(events: list[UsageEventRecord]) -> dict[str, Any]:
|
|
23
|
+
result: dict[str, Any] = {}
|
|
24
|
+
for field in _COUNTERS:
|
|
25
|
+
values = [getattr(event, field) for event in events if getattr(event, field) is not None]
|
|
26
|
+
result[field] = sum(values) if values else None
|
|
27
|
+
workloads = [
|
|
28
|
+
event.input_total_tokens + event.output_total_tokens
|
|
29
|
+
for event in events
|
|
30
|
+
if event.input_total_tokens is not None and event.output_total_tokens is not None
|
|
31
|
+
]
|
|
32
|
+
result.update(
|
|
33
|
+
workload_tokens=sum(workloads) if workloads else None,
|
|
34
|
+
event_count=len(events),
|
|
35
|
+
session_count=len({_session_key(event) for event in events}),
|
|
36
|
+
model_count=len({event.model_name for event in events if event.model_name is not None}),
|
|
37
|
+
incomplete_event_count=sum(
|
|
38
|
+
event.input_total_tokens is None or event.output_total_tokens is None for event in events
|
|
39
|
+
),
|
|
40
|
+
first_seen=min((event.timestamp for event in events), default=None),
|
|
41
|
+
last_seen=max((event.timestamp for event in events), default=None),
|
|
42
|
+
)
|
|
43
|
+
for field in ("first_seen", "last_seen"):
|
|
44
|
+
result[field] = result[field].replace(tzinfo=UTC).isoformat() if result[field] else None
|
|
45
|
+
return result
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def usage_breakdown(events: list[UsageEventRecord]) -> dict[str, Any]:
|
|
49
|
+
providers: dict[str, list[UsageEventRecord]] = defaultdict(list)
|
|
50
|
+
models: dict[tuple[str, str | None], list[UsageEventRecord]] = defaultdict(list)
|
|
51
|
+
sessions: dict[tuple[str, str], list[UsageEventRecord]] = defaultdict(list)
|
|
52
|
+
for event in events:
|
|
53
|
+
providers[event.provider].append(event)
|
|
54
|
+
models[event.provider, event.model_name].append(event)
|
|
55
|
+
sessions[event.provider, _session_key(event)].append(event)
|
|
56
|
+
model_rows = []
|
|
57
|
+
for (provider, model), group in models.items():
|
|
58
|
+
attributions = {event.model_attribution for event in group}
|
|
59
|
+
model_rows.append(dict(
|
|
60
|
+
_summary(group), provider=provider, model_name=model,
|
|
61
|
+
attribution=next(iter(attributions)) if len(attributions) == 1 else "mixed",
|
|
62
|
+
))
|
|
63
|
+
session_rows = []
|
|
64
|
+
for (provider, key), group in sessions.items():
|
|
65
|
+
session_models: dict[str | None, list[UsageEventRecord]] = defaultdict(list)
|
|
66
|
+
for event in group:
|
|
67
|
+
session_models[event.model_name].append(event)
|
|
68
|
+
session_rows.append(dict(
|
|
69
|
+
_summary(group), provider=provider, session_key=key,
|
|
70
|
+
models=[dict(_summary(rows), model_name=model) for model, rows in session_models.items()],
|
|
71
|
+
))
|
|
72
|
+
sort_key = lambda row: (-(row["workload_tokens"] or 0), row["provider"], row.get("model_name") or "")
|
|
73
|
+
return {
|
|
74
|
+
"totals": _summary(events),
|
|
75
|
+
"providers": sorted([dict(_summary(group), provider=provider) for provider, group in providers.items()], key=sort_key),
|
|
76
|
+
"models": sorted(model_rows, key=sort_key),
|
|
77
|
+
"sessions": sorted(session_rows, key=sort_key),
|
|
78
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Dashboard queries over observed, normalized usage only."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from tokenhub.analytics.breakdown import usage_breakdown
|
|
7
|
+
from tokenhub.database.repositories import UsageRepository
|
|
8
|
+
from tokenhub.domain.models import DashboardSummary
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class AnalyticsService:
|
|
12
|
+
def __init__(self, usage_repository: UsageRepository) -> None:
|
|
13
|
+
self.usage_repository = usage_repository
|
|
14
|
+
|
|
15
|
+
def dashboard(self) -> DashboardSummary:
|
|
16
|
+
return self.usage_repository.dashboard_totals()
|
|
17
|
+
|
|
18
|
+
def usage_breakdown(
|
|
19
|
+
self, start: datetime | None = None, end: datetime | None = None
|
|
20
|
+
) -> dict[str, Any]:
|
|
21
|
+
return usage_breakdown(self.usage_repository.observed_events(start, end))
|