tokenhub 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. tokenhub-0.1.0/LICENSE +21 -0
  2. tokenhub-0.1.0/PKG-INFO +230 -0
  3. tokenhub-0.1.0/README.md +213 -0
  4. tokenhub-0.1.0/backend/tokenhub/__init__.py +1 -0
  5. tokenhub-0.1.0/backend/tokenhub/__main__.py +8 -0
  6. tokenhub-0.1.0/backend/tokenhub/analytics/__init__.py +1 -0
  7. tokenhub-0.1.0/backend/tokenhub/analytics/breakdown.py +78 -0
  8. tokenhub-0.1.0/backend/tokenhub/analytics/service.py +21 -0
  9. tokenhub-0.1.0/backend/tokenhub/api/__init__.py +7 -0
  10. tokenhub-0.1.0/backend/tokenhub/api/container.py +146 -0
  11. tokenhub-0.1.0/backend/tokenhub/api/routes.py +267 -0
  12. tokenhub-0.1.0/backend/tokenhub/app.py +77 -0
  13. tokenhub-0.1.0/backend/tokenhub/cli.py +65 -0
  14. tokenhub-0.1.0/backend/tokenhub/connectors/__init__.py +1 -0
  15. tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/__init__.py +3 -0
  16. tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/connector.py +107 -0
  17. tokenhub-0.1.0/backend/tokenhub/connectors/antigravity/parser.py +238 -0
  18. tokenhub-0.1.0/backend/tokenhub/connectors/claude/__init__.py +3 -0
  19. tokenhub-0.1.0/backend/tokenhub/connectors/claude/connector.py +152 -0
  20. tokenhub-0.1.0/backend/tokenhub/connectors/claude/parser.py +182 -0
  21. tokenhub-0.1.0/backend/tokenhub/connectors/codex/__init__.py +3 -0
  22. tokenhub-0.1.0/backend/tokenhub/connectors/codex/connector.py +209 -0
  23. tokenhub-0.1.0/backend/tokenhub/connectors/codex/parser.py +261 -0
  24. tokenhub-0.1.0/backend/tokenhub/connectors/copilot/__init__.py +3 -0
  25. tokenhub-0.1.0/backend/tokenhub/connectors/copilot/connector.py +155 -0
  26. tokenhub-0.1.0/backend/tokenhub/connectors/copilot/parser.py +224 -0
  27. tokenhub-0.1.0/backend/tokenhub/connectors/hermes/__init__.py +3 -0
  28. tokenhub-0.1.0/backend/tokenhub/connectors/hermes/connector.py +106 -0
  29. tokenhub-0.1.0/backend/tokenhub/connectors/hermes/parser.py +246 -0
  30. tokenhub-0.1.0/backend/tokenhub/connectors/metadata.py +11 -0
  31. tokenhub-0.1.0/backend/tokenhub/connectors/protocol.py +116 -0
  32. tokenhub-0.1.0/backend/tokenhub/connectors/registry.py +42 -0
  33. tokenhub-0.1.0/backend/tokenhub/database/__init__.py +11 -0
  34. tokenhub-0.1.0/backend/tokenhub/database/migrations/env.py +58 -0
  35. tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0001_initial.py +85 -0
  36. tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0002_cursor_prefix_fingerprint.py +29 -0
  37. tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0003_source_trust_and_quality.py +44 -0
  38. tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0004_auto_import_roots.py +25 -0
  39. tokenhub-0.1.0/backend/tokenhub/database/migrations/versions/0005_usage_metadata.py +24 -0
  40. tokenhub-0.1.0/backend/tokenhub/database/migrations.py +47 -0
  41. tokenhub-0.1.0/backend/tokenhub/database/models.py +87 -0
  42. tokenhub-0.1.0/backend/tokenhub/database/repositories.py +491 -0
  43. tokenhub-0.1.0/backend/tokenhub/database/session.py +35 -0
  44. tokenhub-0.1.0/backend/tokenhub/discovery/__init__.py +1 -0
  45. tokenhub-0.1.0/backend/tokenhub/discovery/service.py +51 -0
  46. tokenhub-0.1.0/backend/tokenhub/domain/__init__.py +17 -0
  47. tokenhub-0.1.0/backend/tokenhub/domain/models.py +273 -0
  48. tokenhub-0.1.0/backend/tokenhub/ingestion/__init__.py +1 -0
  49. tokenhub-0.1.0/backend/tokenhub/ingestion/collection.py +116 -0
  50. tokenhub-0.1.0/backend/tokenhub/ingestion/service.py +137 -0
  51. tokenhub-0.1.0/backend/tokenhub/runtime/__init__.py +5 -0
  52. tokenhub-0.1.0/backend/tokenhub/runtime/client.py +69 -0
  53. tokenhub-0.1.0/backend/tokenhub/runtime/control.py +62 -0
  54. tokenhub-0.1.0/backend/tokenhub/runtime/instance.py +79 -0
  55. tokenhub-0.1.0/backend/tokenhub/runtime/manager.py +161 -0
  56. tokenhub-0.1.0/backend/tokenhub/security/__init__.py +6 -0
  57. tokenhub-0.1.0/backend/tokenhub/security/http.py +108 -0
  58. tokenhub-0.1.0/backend/tokenhub/security/paths.py +204 -0
  59. tokenhub-0.1.0/backend/tokenhub/security/redaction.py +13 -0
  60. tokenhub-0.1.0/backend/tokenhub/security/windows_paths.py +223 -0
  61. tokenhub-0.1.0/backend/tokenhub/server.py +29 -0
  62. tokenhub-0.1.0/backend/tokenhub/settings.py +53 -0
  63. tokenhub-0.1.0/backend/tokenhub/web/assets/index-B2a5VymX.css +1 -0
  64. tokenhub-0.1.0/backend/tokenhub/web/assets/index-Ctyi0WkJ.js +40 -0
  65. tokenhub-0.1.0/backend/tokenhub/web/favicon.svg +4 -0
  66. tokenhub-0.1.0/backend/tokenhub/web/index.html +16 -0
  67. tokenhub-0.1.0/backend/tokenhub.egg-info/PKG-INFO +230 -0
  68. tokenhub-0.1.0/backend/tokenhub.egg-info/SOURCES.txt +73 -0
  69. tokenhub-0.1.0/backend/tokenhub.egg-info/dependency_links.txt +1 -0
  70. tokenhub-0.1.0/backend/tokenhub.egg-info/entry_points.txt +2 -0
  71. tokenhub-0.1.0/backend/tokenhub.egg-info/requires.txt +7 -0
  72. tokenhub-0.1.0/backend/tokenhub.egg-info/top_level.txt +1 -0
  73. tokenhub-0.1.0/pyproject.toml +52 -0
  74. tokenhub-0.1.0/setup.cfg +4 -0
  75. tokenhub-0.1.0/tests/test_package.py +5 -0
tokenhub-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Aldrin Joseph
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,230 @@
1
+ Metadata-Version: 2.4
2
+ Name: tokenhub
3
+ Version: 0.1.0
4
+ Summary: Local-first TokenHub
5
+ License-Expression: MIT
6
+ Requires-Python: >=3.12
7
+ Description-Content-Type: text/markdown
8
+ License-File: LICENSE
9
+ Requires-Dist: fastapi
10
+ Requires-Dist: uvicorn[standard]
11
+ Requires-Dist: pydantic
12
+ Requires-Dist: sqlalchemy
13
+ Requires-Dist: alembic
14
+ Requires-Dist: filelock<4,>=3.12
15
+ Requires-Dist: psutil<8,>=5.9
16
+ Dynamic: license-file
17
+
18
+ # TokenHub
19
+
20
+ A local-first usage observatory. TokenHub detects the coding agents installed on
21
+ this machine, imports usage from sources you explicitly approve, and
22
+ shows agent, model, and session usage in a local explorer. Nothing leaves the machine: no provider
23
+ API is called, no provider executable is launched, and only usage metadata is retained. Credentials and provider accounts are never
24
+ read; transcript content is never stored or shown. Approved session files can contain chat text, which the parsers discard.
25
+
26
+ Supported sources: **OpenAI Codex** session usage, **Claude Code** project and
27
+ subagent session logs, **Hermes Agent** session counters in its local state
28
+ database, **VS Code Copilot Chat** saved sessions, and **Antigravity** conversation
29
+ databases.
30
+
31
+ ## Install and run
32
+
33
+ Requires Python 3.12 or newer. After the public package is released:
34
+
35
+ ```bash
36
+ pip install tokenhub
37
+ tokenhub # starts in the background and opens the local dashboard
38
+ tokenhub status # prints its local URL
39
+ tokenhub open # opens an already running dashboard
40
+ tokenhub stop # stops this TokenHub instance
41
+ ```
42
+
43
+ `tokenhub start --no-open` starts without opening a browser. Use
44
+ `tokenhub start --port 9000` to choose another loopback port. A second start
45
+ reuses the running instance, including its chosen port. The server continues
46
+ after the terminal closes; run `tokenhub` again after a reboot or login. Logs
47
+ and local data are stored in `~/.tokenhub/`, with startup details in
48
+ `~/.tokenhub/runtime.log`.
49
+
50
+ For development from this checkout, run `uv sync --all-groups` and
51
+ `uv run tokenhub --no-open`.
52
+
53
+ TokenHub binds to `127.0.0.1:7432` by default and prints that URL when it
54
+ starts. It is loopback-only by design:
55
+
56
+ - the bind host is validated at startup — `0.0.0.0`, `::`, or any non-loopback
57
+ value raises rather than serving;
58
+ - every request must carry a loopback `Host` header (`localhost`, `127.0.0.1`,
59
+ `[::1]`, with the configured port), so a DNS-rebinding page cannot reach the
60
+ API even from the local browser;
61
+ - state-changing requests (`approve`, `rescan`, `rebuild`) must also carry the
62
+ request's own origin; otherwise they are refused with `403`;
63
+ - no CORS headers are ever added, and `X-Forwarded-*` headers are ignored.
64
+
65
+ The UI is bundled with the Python package. To refresh it while developing:
66
+
67
+ ```bash
68
+ npm --prefix frontend ci
69
+ npm --prefix frontend run build # writes backend/tokenhub/web
70
+ ```
71
+
72
+ The packaged UI is mounted at `/`; the API also works if the built UI is absent.
73
+
74
+ ## Automatic collection
75
+
76
+ While TokenHub is running, it scans approved sources from all five providers at startup and every
77
+ 30 seconds. The explorer refreshes itself every 10 seconds. Approving a provider
78
+ in the UI also starts its first import immediately. Unchanged session files are skipped. Hermes and Antigravity databases are checked on each scan
79
+ because active usage can live in their SQLite journal. Repeated scans do not
80
+ increase totals. Automatic scans of unchanged data do not create extra
81
+ import-history rows, including after restarting TokenHub.
82
+
83
+ Choose **Approve and import** on a provider card to approve its discovered
84
+ usage folder once and include both existing and future sessions automatically.
85
+ This consent survives restarts and is pinned to the folder's device and inode;
86
+ replacing the folder or redirecting it through a symlink cannot authorize another
87
+ location. **Stop including new sessions** removes that folder consent; previously
88
+ approved files keep updating. Without folder consent, new files require approval.
89
+
90
+ `GET /api/v1/collection` reports the scan interval, folder-consent state, last
91
+ completed scan, and failed-source count. Same-origin `POST` requests to
92
+ `/api/v1/collection/{provider}/enable` and `/disable` change consent (`codex`,
93
+ `claude_code`, `hermes`, `vscode_copilot`, or `antigravity`). Responses contain
94
+ no private paths. Failed sources are retried without preventing other sources
95
+ from updating, and collection runs even when the dashboard is closed.
96
+
97
+ For Codex, only per-response `token_usage_record.payload.usage` values contribute
98
+ to totals.
99
+ Known session messages are skipped. Cumulative token snapshots, malformed records,
100
+ and unknown usage structures remain visible as unsupported records; cumulative
101
+ snapshots are excluded from totals to avoid double counting. Workload is
102
+ input plus output, with cached input and reasoning already included in those
103
+ counts. Token counts are not billing amounts. The UI uses K/M/B abbreviations
104
+ and shows the exact count on hover.
105
+
106
+ Claude Code reads `message.usage` from assistant records under its `projects`
107
+ folder. Repeated streaming chunks are consolidated by message ID, copied history in
108
+ resumed/forked session files is counted once, and later usage updates replace
109
+ the session's previous normalized records. Hermes reads
110
+ only allowlisted session identifiers, timestamps, and token counters from a
111
+ no-follow snapshot of `state.db` and its journal; no provider database is written.
112
+ Each Hermes session contributes once, with running totals replaced atomically as
113
+ they change. Cache reads and writes are included in the input total for Claude
114
+ and Hermes because both store them in separate buckets. Hermes session
115
+ contributions are marked `high` quality; Claude message usage is marked `exact`.
116
+
117
+ VS Code Copilot Chat imports saved `chatSessions/*.jsonl` requests from VS Code's
118
+ workspace storage. It records each Copilot request's model, session, input tokens,
119
+ and output tokens when VS Code saved those counters. Later session patches replace
120
+ earlier counters; unchanged files do not add usage again. Requests without token
121
+ counters remain unsupported rather than receiving estimated values. Cache and
122
+ reasoning counters are not available in these saved requests. Inline suggestions,
123
+ Copilot CLI, and Copilot activity in other editors are outside this source's scope.
124
+ `VSCODE_USER_DATA_DIR` can point to a different VS Code user-data directory,
125
+ such as an Insiders installation.
126
+
127
+ Antigravity imports per-generation model, timestamp, input, cache-read, output,
128
+ and reasoning counters from approved `~/.gemini/antigravity/conversations/*.db`
129
+ files. It reads a snapshot of the database and its active journal without writing
130
+ to the provider database. Antigravity's local metadata format is private, so
131
+ unrecognized or inconsistent generations are excluded and reported as partial
132
+ instead of estimated. `ANTIGRAVITY_HOME` can point to another Antigravity root.
133
+
134
+ ## Discovery and approval
135
+
136
+ `GET /api/v1/discovery` reports each provider's display name, connection state,
137
+ a confidence level derived from its evidence (`high` when two or more
138
+ independent signals were found, `medium` for one, `low` for none), the evidence
139
+ codes it found (`executable_on_path`, `known_root_exists`, `configuration_found`,
140
+ `session_source_found`, `state_database_found`), and a stable path fingerprint.
141
+ Discovery reads **presence only** — it stats candidate roots and directories and
142
+ never opens a provider file.
143
+
144
+ The provider card combines approval and first import in one click. It also
145
+ enables future session imports for that provider. The lower-level API remains
146
+ available for one-source troubleshooting:
147
+
148
+ 1. `POST /api/v1/sources/{source_id}/approve` persists the canonical path for
149
+ that one source. Before approval, responses contain no absolute path at all.
150
+ 2. `POST /api/v1/sources/{source_id}/rescan` imports new records through the
151
+ connector's parser. Repeating a rescan over unchanged bytes inserts nothing;
152
+ `POST /api/v1/rebuild` re-derives all normalized events from scratch and
153
+ leaves totals unchanged.
154
+
155
+ `GET /api/v1/dashboard` and `GET /api/v1/data-quality` report observed usage.
156
+ Workload is input total plus output total; cache and reasoning are breakdowns,
157
+ never extra usage. A metric the data cannot support is returned as `null` and
158
+ rendered as an em dash — TokenHub never substitutes `0` for "unknown".
159
+
160
+ The default **Usage explorer** combines the consolidated token summary with
161
+ agent, model, and session breakdowns for Codex, Claude Code, Hermes Agent,
162
+ VS Code Copilot, and Antigravity. Filter by all time, today, yesterday, the last
163
+ 7 or 30 days, or a selected local calendar date.
164
+ Choose an agent to rank its models by total, input, output, cache-read, or
165
+ reasoning tokens. Search for a model, select it to see its sessions, and expand
166
+ a session to compare the models used inside it. Exact counts are available on
167
+ hover. Source health and import quality appear together under **Local sources**.
168
+
169
+ `GET /api/v1/usage` returns canonical totals plus agent, model, and session
170
+ breakdowns. Session identifiers are opaque, stable hashes; original session IDs
171
+ and source paths are not exposed. The same Claude message deduplication is used
172
+ throughout the explorer. Optional timezone-aware `from` and `to` query bounds
173
+ filter usage by recorded event time; the upper bound is exclusive. Known older
174
+ parsers re-read approved sources on startup to add
175
+ model/session metadata without changing previously imported Codex counters.
176
+
177
+ ## Data boundaries
178
+
179
+ - Only usage metadata is normalized. Prompts, messages, tool output, transcript
180
+ bodies, and raw provider records are never retained or returned by the API.
181
+ - Provider credentials, API keys, `auth.json`, keychain, and cookie stores are
182
+ never read. Approved session JSONL can contain chat text; only allowlisted
183
+ usage fields are normalized, and raw content is never stored or returned.
184
+ - Anything outside an approved source's own approved root: canonical-path
185
+ validation rejects symlink escapes, and the approved root is re-validated by
186
+ device/inode at every scan.
187
+
188
+ ## Tests
189
+
190
+ ```bash
191
+ uv run pytest -q # unit + API
192
+ uv run pytest tests/integration -m integration -q # real app stack, synthetic homes
193
+ uv run ruff check backend tests
194
+ uv run mypy backend
195
+ cd frontend && npm run test -- --run && npm run build
196
+ uv build
197
+ python scripts/verify_distribution.py dist/* # tests installed wheel and source archive
198
+ ```
199
+
200
+ Every test is offline and deterministic: synthetic fixtures in a temp directory,
201
+ no real home directory, no network, no provider subprocess. A test that points
202
+ at a real provider path or opens a real credential store is a bug.
203
+
204
+ ## Layout
205
+
206
+ | Path | Responsibility |
207
+ |---|---|
208
+ | `backend/tokenhub/domain/` | provider-neutral enums and data models |
209
+ | `backend/tokenhub/security/` | path containment, redaction, Host/Origin policy |
210
+ | `backend/tokenhub/connectors/` | one module per provider behind a shared contract |
211
+ | `backend/tokenhub/discovery/` | presence discovery and the pre-approval handoff |
212
+ | `backend/tokenhub/ingestion/` | approval gate, incremental scan, rebuild |
213
+ | `backend/tokenhub/database/` | SQLAlchemy models, repositories, Alembic migrations |
214
+ | `backend/tokenhub/analytics/` | dashboard aggregates over observed deltas |
215
+ | `backend/tokenhub/api/` | versioned `/api/v1` routes and the app container |
216
+ | `frontend/` | React + TypeScript + Vite client (Vitest, MSW) |
217
+
218
+ ## Current limitations
219
+
220
+ - **Local schemas only.** Unknown usage structures and malformed records are
221
+ reported as partial or unavailable. Hermes needs a `sessions` table containing
222
+ session IDs, start timestamps, and input/output counters. An unsupported or
223
+ unreadable database leaves previously imported usage intact.
224
+ - **No cost or pricing.** TokenHub reports token counts only — no currency, no
225
+ provider plan data.
226
+ - **Model attribution depends on recorded metadata.** Missing model names appear
227
+ in “Model not recorded”. Hermes exposes session totals with a reported model;
228
+ model switches within that session cannot be split by its current counters.
229
+ - **No trend chart.** Date filters use observed usage timestamps. TokenHub does
230
+ not guess daily allocations for counters that lack event dates.
@@ -0,0 +1,213 @@
1
+ # TokenHub
2
+
3
+ A local-first usage observatory. TokenHub detects the coding agents installed on
4
+ this machine, imports usage from sources you explicitly approve, and
5
+ shows agent, model, and session usage in a local explorer. Nothing leaves the machine: no provider
6
+ API is called, no provider executable is launched, and only usage metadata is retained. Credentials and provider accounts are never
7
+ read; transcript content is never stored or shown. Approved session files can contain chat text, which the parsers discard.
8
+
9
+ Supported sources: **OpenAI Codex** session usage, **Claude Code** project and
10
+ subagent session logs, **Hermes Agent** session counters in its local state
11
+ database, **VS Code Copilot Chat** saved sessions, and **Antigravity** conversation
12
+ databases.
13
+
14
+ ## Install and run
15
+
16
+ Requires Python 3.12 or newer. After the public package is released:
17
+
18
+ ```bash
19
+ pip install tokenhub
20
+ tokenhub # starts in the background and opens the local dashboard
21
+ tokenhub status # prints its local URL
22
+ tokenhub open # opens an already running dashboard
23
+ tokenhub stop # stops this TokenHub instance
24
+ ```
25
+
26
+ `tokenhub start --no-open` starts without opening a browser. Use
27
+ `tokenhub start --port 9000` to choose another loopback port. A second start
28
+ reuses the running instance, including its chosen port. The server continues
29
+ after the terminal closes; run `tokenhub` again after a reboot or login. Logs
30
+ and local data are stored in `~/.tokenhub/`, with startup details in
31
+ `~/.tokenhub/runtime.log`.
32
+
33
+ For development from this checkout, run `uv sync --all-groups` and
34
+ `uv run tokenhub --no-open`.
35
+
36
+ TokenHub binds to `127.0.0.1:7432` by default and prints that URL when it
37
+ starts. It is loopback-only by design:
38
+
39
+ - the bind host is validated at startup — `0.0.0.0`, `::`, or any non-loopback
40
+ value raises rather than serving;
41
+ - every request must carry a loopback `Host` header (`localhost`, `127.0.0.1`,
42
+ `[::1]`, with the configured port), so a DNS-rebinding page cannot reach the
43
+ API even from the local browser;
44
+ - state-changing requests (`approve`, `rescan`, `rebuild`) must also carry the
45
+ request's own origin; otherwise they are refused with `403`;
46
+ - no CORS headers are ever added, and `X-Forwarded-*` headers are ignored.
47
+
48
+ The UI is bundled with the Python package. To refresh it while developing:
49
+
50
+ ```bash
51
+ npm --prefix frontend ci
52
+ npm --prefix frontend run build # writes backend/tokenhub/web
53
+ ```
54
+
55
+ The packaged UI is mounted at `/`; the API also works if the built UI is absent.
56
+
57
+ ## Automatic collection
58
+
59
+ While TokenHub is running, it scans approved sources from all five providers at startup and every
60
+ 30 seconds. The explorer refreshes itself every 10 seconds. Approving a provider
61
+ in the UI also starts its first import immediately. Unchanged session files are skipped. Hermes and Antigravity databases are checked on each scan
62
+ because active usage can live in their SQLite journal. Repeated scans do not
63
+ increase totals. Automatic scans of unchanged data do not create extra
64
+ import-history rows, including after restarting TokenHub.
65
+
66
+ Choose **Approve and import** on a provider card to approve its discovered
67
+ usage folder once and include both existing and future sessions automatically.
68
+ This consent survives restarts and is pinned to the folder's device and inode;
69
+ replacing the folder or redirecting it through a symlink cannot authorize another
70
+ location. **Stop including new sessions** removes that folder consent; previously
71
+ approved files keep updating. Without folder consent, new files require approval.
72
+
73
+ `GET /api/v1/collection` reports the scan interval, folder-consent state, last
74
+ completed scan, and failed-source count. Same-origin `POST` requests to
75
+ `/api/v1/collection/{provider}/enable` and `/disable` change consent (`codex`,
76
+ `claude_code`, `hermes`, `vscode_copilot`, or `antigravity`). Responses contain
77
+ no private paths. Failed sources are retried without preventing other sources
78
+ from updating, and collection runs even when the dashboard is closed.
79
+
80
+ For Codex, only per-response `token_usage_record.payload.usage` values contribute
81
+ to totals.
82
+ Known session messages are skipped. Cumulative token snapshots, malformed records,
83
+ and unknown usage structures remain visible as unsupported records; cumulative
84
+ snapshots are excluded from totals to avoid double counting. Workload is
85
+ input plus output, with cached input and reasoning already included in those
86
+ counts. Token counts are not billing amounts. The UI uses K/M/B abbreviations
87
+ and shows the exact count on hover.
88
+
89
+ Claude Code reads `message.usage` from assistant records under its `projects`
90
+ folder. Repeated streaming chunks are consolidated by message ID, copied history in
91
+ resumed/forked session files is counted once, and later usage updates replace
92
+ the session's previous normalized records. Hermes reads
93
+ only allowlisted session identifiers, timestamps, and token counters from a
94
+ no-follow snapshot of `state.db` and its journal; no provider database is written.
95
+ Each Hermes session contributes once, with running totals replaced atomically as
96
+ they change. Cache reads and writes are included in the input total for Claude
97
+ and Hermes because both store them in separate buckets. Hermes session
98
+ contributions are marked `high` quality; Claude message usage is marked `exact`.
99
+
100
+ VS Code Copilot Chat imports saved `chatSessions/*.jsonl` requests from VS Code's
101
+ workspace storage. It records each Copilot request's model, session, input tokens,
102
+ and output tokens when VS Code saved those counters. Later session patches replace
103
+ earlier counters; unchanged files do not add usage again. Requests without token
104
+ counters remain unsupported rather than receiving estimated values. Cache and
105
+ reasoning counters are not available in these saved requests. Inline suggestions,
106
+ Copilot CLI, and Copilot activity in other editors are outside this source's scope.
107
+ `VSCODE_USER_DATA_DIR` can point to a different VS Code user-data directory,
108
+ such as an Insiders installation.
109
+
110
+ Antigravity imports per-generation model, timestamp, input, cache-read, output,
111
+ and reasoning counters from approved `~/.gemini/antigravity/conversations/*.db`
112
+ files. It reads a snapshot of the database and its active journal without writing
113
+ to the provider database. Antigravity's local metadata format is private, so
114
+ unrecognized or inconsistent generations are excluded and reported as partial
115
+ instead of estimated. `ANTIGRAVITY_HOME` can point to another Antigravity root.
116
+
117
+ ## Discovery and approval
118
+
119
+ `GET /api/v1/discovery` reports each provider's display name, connection state,
120
+ a confidence level derived from its evidence (`high` when two or more
121
+ independent signals were found, `medium` for one, `low` for none), the evidence
122
+ codes it found (`executable_on_path`, `known_root_exists`, `configuration_found`,
123
+ `session_source_found`, `state_database_found`), and a stable path fingerprint.
124
+ Discovery reads **presence only** — it stats candidate roots and directories and
125
+ never opens a provider file.
126
+
127
+ The provider card combines approval and first import in one click. It also
128
+ enables future session imports for that provider. The lower-level API remains
129
+ available for one-source troubleshooting:
130
+
131
+ 1. `POST /api/v1/sources/{source_id}/approve` persists the canonical path for
132
+ that one source. Before approval, responses contain no absolute path at all.
133
+ 2. `POST /api/v1/sources/{source_id}/rescan` imports new records through the
134
+ connector's parser. Repeating a rescan over unchanged bytes inserts nothing;
135
+ `POST /api/v1/rebuild` re-derives all normalized events from scratch and
136
+ leaves totals unchanged.
137
+
138
+ `GET /api/v1/dashboard` and `GET /api/v1/data-quality` report observed usage.
139
+ Workload is input total plus output total; cache and reasoning are breakdowns,
140
+ never extra usage. A metric the data cannot support is returned as `null` and
141
+ rendered as an em dash — TokenHub never substitutes `0` for "unknown".
142
+
143
+ The default **Usage explorer** combines the consolidated token summary with
144
+ agent, model, and session breakdowns for Codex, Claude Code, Hermes Agent,
145
+ VS Code Copilot, and Antigravity. Filter by all time, today, yesterday, the last
146
+ 7 or 30 days, or a selected local calendar date.
147
+ Choose an agent to rank its models by total, input, output, cache-read, or
148
+ reasoning tokens. Search for a model, select it to see its sessions, and expand
149
+ a session to compare the models used inside it. Exact counts are available on
150
+ hover. Source health and import quality appear together under **Local sources**.
151
+
152
+ `GET /api/v1/usage` returns canonical totals plus agent, model, and session
153
+ breakdowns. Session identifiers are opaque, stable hashes; original session IDs
154
+ and source paths are not exposed. The same Claude message deduplication is used
155
+ throughout the explorer. Optional timezone-aware `from` and `to` query bounds
156
+ filter usage by recorded event time; the upper bound is exclusive. Known older
157
+ parsers re-read approved sources on startup to add
158
+ model/session metadata without changing previously imported Codex counters.
159
+
160
+ ## Data boundaries
161
+
162
+ - Only usage metadata is normalized. Prompts, messages, tool output, transcript
163
+ bodies, and raw provider records are never retained or returned by the API.
164
+ - Provider credentials, API keys, `auth.json`, keychain, and cookie stores are
165
+ never read. Approved session JSONL can contain chat text; only allowlisted
166
+ usage fields are normalized, and raw content is never stored or returned.
167
+ - Anything outside an approved source's own approved root: canonical-path
168
+ validation rejects symlink escapes, and the approved root is re-validated by
169
+ device/inode at every scan.
170
+
171
+ ## Tests
172
+
173
+ ```bash
174
+ uv run pytest -q # unit + API
175
+ uv run pytest tests/integration -m integration -q # real app stack, synthetic homes
176
+ uv run ruff check backend tests
177
+ uv run mypy backend
178
+ cd frontend && npm run test -- --run && npm run build
179
+ uv build
180
+ python scripts/verify_distribution.py dist/* # tests installed wheel and source archive
181
+ ```
182
+
183
+ Every test is offline and deterministic: synthetic fixtures in a temp directory,
184
+ no real home directory, no network, no provider subprocess. A test that points
185
+ at a real provider path or opens a real credential store is a bug.
186
+
187
+ ## Layout
188
+
189
+ | Path | Responsibility |
190
+ |---|---|
191
+ | `backend/tokenhub/domain/` | provider-neutral enums and data models |
192
+ | `backend/tokenhub/security/` | path containment, redaction, Host/Origin policy |
193
+ | `backend/tokenhub/connectors/` | one module per provider behind a shared contract |
194
+ | `backend/tokenhub/discovery/` | presence discovery and the pre-approval handoff |
195
+ | `backend/tokenhub/ingestion/` | approval gate, incremental scan, rebuild |
196
+ | `backend/tokenhub/database/` | SQLAlchemy models, repositories, Alembic migrations |
197
+ | `backend/tokenhub/analytics/` | dashboard aggregates over observed deltas |
198
+ | `backend/tokenhub/api/` | versioned `/api/v1` routes and the app container |
199
+ | `frontend/` | React + TypeScript + Vite client (Vitest, MSW) |
200
+
201
+ ## Current limitations
202
+
203
+ - **Local schemas only.** Unknown usage structures and malformed records are
204
+ reported as partial or unavailable. Hermes needs a `sessions` table containing
205
+ session IDs, start timestamps, and input/output counters. An unsupported or
206
+ unreadable database leaves previously imported usage intact.
207
+ - **No cost or pricing.** TokenHub reports token counts only — no currency, no
208
+ provider plan data.
209
+ - **Model attribution depends on recorded metadata.** Missing model names appear
210
+ in “Model not recorded”. Hermes exposes session totals with a reported model;
211
+ model switches within that session cannot be split by its current counters.
212
+ - **No trend chart.** Date filters use observed usage timestamps. TokenHub does
213
+ not guess daily allocations for counters that lack event dates.
@@ -0,0 +1 @@
1
+ """TokenHub package."""
@@ -0,0 +1,8 @@
1
+ """Module entrypoint for the detached child and interactive command."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from tokenhub.cli import main
6
+
7
+ if __name__ == "__main__":
8
+ raise SystemExit(main())
@@ -0,0 +1 @@
1
+ """Provider-neutral analytics over normalized local events."""
@@ -0,0 +1,78 @@
1
+ """Agent, model, and session views of the canonical observed workload."""
2
+
3
+ import hashlib
4
+ from collections import defaultdict
5
+ from datetime import UTC
6
+ from typing import Any
7
+
8
+ from tokenhub.database.models import UsageEventRecord
9
+
10
+ _COUNTERS = (
11
+ "input_total_tokens", "output_total_tokens", "cache_read_tokens",
12
+ "cache_write_tokens", "reasoning_tokens",
13
+ )
14
+
15
+
16
+ def _session_key(event: UsageEventRecord) -> str:
17
+ # Stable across refreshes, with no original session id or filesystem path.
18
+ identity = f"{event.connector_id}\0{event.session_id or event.source_id}"
19
+ return hashlib.sha256(identity.encode()).hexdigest()[:24]
20
+
21
+
22
+ def _summary(events: list[UsageEventRecord]) -> dict[str, Any]:
23
+ result: dict[str, Any] = {}
24
+ for field in _COUNTERS:
25
+ values = [getattr(event, field) for event in events if getattr(event, field) is not None]
26
+ result[field] = sum(values) if values else None
27
+ workloads = [
28
+ event.input_total_tokens + event.output_total_tokens
29
+ for event in events
30
+ if event.input_total_tokens is not None and event.output_total_tokens is not None
31
+ ]
32
+ result.update(
33
+ workload_tokens=sum(workloads) if workloads else None,
34
+ event_count=len(events),
35
+ session_count=len({_session_key(event) for event in events}),
36
+ model_count=len({event.model_name for event in events if event.model_name is not None}),
37
+ incomplete_event_count=sum(
38
+ event.input_total_tokens is None or event.output_total_tokens is None for event in events
39
+ ),
40
+ first_seen=min((event.timestamp for event in events), default=None),
41
+ last_seen=max((event.timestamp for event in events), default=None),
42
+ )
43
+ for field in ("first_seen", "last_seen"):
44
+ result[field] = result[field].replace(tzinfo=UTC).isoformat() if result[field] else None
45
+ return result
46
+
47
+
48
+ def usage_breakdown(events: list[UsageEventRecord]) -> dict[str, Any]:
49
+ providers: dict[str, list[UsageEventRecord]] = defaultdict(list)
50
+ models: dict[tuple[str, str | None], list[UsageEventRecord]] = defaultdict(list)
51
+ sessions: dict[tuple[str, str], list[UsageEventRecord]] = defaultdict(list)
52
+ for event in events:
53
+ providers[event.provider].append(event)
54
+ models[event.provider, event.model_name].append(event)
55
+ sessions[event.provider, _session_key(event)].append(event)
56
+ model_rows = []
57
+ for (provider, model), group in models.items():
58
+ attributions = {event.model_attribution for event in group}
59
+ model_rows.append(dict(
60
+ _summary(group), provider=provider, model_name=model,
61
+ attribution=next(iter(attributions)) if len(attributions) == 1 else "mixed",
62
+ ))
63
+ session_rows = []
64
+ for (provider, key), group in sessions.items():
65
+ session_models: dict[str | None, list[UsageEventRecord]] = defaultdict(list)
66
+ for event in group:
67
+ session_models[event.model_name].append(event)
68
+ session_rows.append(dict(
69
+ _summary(group), provider=provider, session_key=key,
70
+ models=[dict(_summary(rows), model_name=model) for model, rows in session_models.items()],
71
+ ))
72
+ sort_key = lambda row: (-(row["workload_tokens"] or 0), row["provider"], row.get("model_name") or "")
73
+ return {
74
+ "totals": _summary(events),
75
+ "providers": sorted([dict(_summary(group), provider=provider) for provider, group in providers.items()], key=sort_key),
76
+ "models": sorted(model_rows, key=sort_key),
77
+ "sessions": sorted(session_rows, key=sort_key),
78
+ }
@@ -0,0 +1,21 @@
1
+ """Dashboard queries over observed, normalized usage only."""
2
+
3
+ from datetime import datetime
4
+ from typing import Any
5
+
6
+ from tokenhub.analytics.breakdown import usage_breakdown
7
+ from tokenhub.database.repositories import UsageRepository
8
+ from tokenhub.domain.models import DashboardSummary
9
+
10
+
11
+ class AnalyticsService:
12
+ def __init__(self, usage_repository: UsageRepository) -> None:
13
+ self.usage_repository = usage_repository
14
+
15
+ def dashboard(self) -> DashboardSummary:
16
+ return self.usage_repository.dashboard_totals()
17
+
18
+ def usage_breakdown(
19
+ self, start: datetime | None = None, end: datetime | None = None
20
+ ) -> dict[str, Any]:
21
+ return usage_breakdown(self.usage_repository.observed_events(start, end))
@@ -0,0 +1,7 @@
1
+ """HTTP surface for the local TokenHub server.
2
+
3
+ * :mod:`tokenhub.api.container` — startup/shutdown and shared services
4
+ * :mod:`tokenhub.api.routes` — the versioned `/api/v1` routes
5
+
6
+ The application factory lives in :mod:`tokenhub.app`.
7
+ """