tokenhub 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokenhub/__init__.py +1 -0
- tokenhub/__main__.py +8 -0
- tokenhub/analytics/__init__.py +1 -0
- tokenhub/analytics/breakdown.py +78 -0
- tokenhub/analytics/service.py +21 -0
- tokenhub/api/__init__.py +7 -0
- tokenhub/api/container.py +146 -0
- tokenhub/api/routes.py +267 -0
- tokenhub/app.py +77 -0
- tokenhub/cli.py +65 -0
- tokenhub/connectors/__init__.py +1 -0
- tokenhub/connectors/antigravity/__init__.py +3 -0
- tokenhub/connectors/antigravity/connector.py +107 -0
- tokenhub/connectors/antigravity/parser.py +238 -0
- tokenhub/connectors/claude/__init__.py +3 -0
- tokenhub/connectors/claude/connector.py +152 -0
- tokenhub/connectors/claude/parser.py +182 -0
- tokenhub/connectors/codex/__init__.py +3 -0
- tokenhub/connectors/codex/connector.py +209 -0
- tokenhub/connectors/codex/parser.py +261 -0
- tokenhub/connectors/copilot/__init__.py +3 -0
- tokenhub/connectors/copilot/connector.py +155 -0
- tokenhub/connectors/copilot/parser.py +224 -0
- tokenhub/connectors/hermes/__init__.py +3 -0
- tokenhub/connectors/hermes/connector.py +106 -0
- tokenhub/connectors/hermes/parser.py +246 -0
- tokenhub/connectors/metadata.py +11 -0
- tokenhub/connectors/protocol.py +116 -0
- tokenhub/connectors/registry.py +42 -0
- tokenhub/database/__init__.py +11 -0
- tokenhub/database/migrations/env.py +58 -0
- tokenhub/database/migrations/versions/0001_initial.py +85 -0
- tokenhub/database/migrations/versions/0002_cursor_prefix_fingerprint.py +29 -0
- tokenhub/database/migrations/versions/0003_source_trust_and_quality.py +44 -0
- tokenhub/database/migrations/versions/0004_auto_import_roots.py +25 -0
- tokenhub/database/migrations/versions/0005_usage_metadata.py +24 -0
- tokenhub/database/migrations.py +47 -0
- tokenhub/database/models.py +87 -0
- tokenhub/database/repositories.py +491 -0
- tokenhub/database/session.py +35 -0
- tokenhub/discovery/__init__.py +1 -0
- tokenhub/discovery/service.py +51 -0
- tokenhub/domain/__init__.py +17 -0
- tokenhub/domain/models.py +273 -0
- tokenhub/ingestion/__init__.py +1 -0
- tokenhub/ingestion/collection.py +116 -0
- tokenhub/ingestion/service.py +137 -0
- tokenhub/runtime/__init__.py +5 -0
- tokenhub/runtime/client.py +69 -0
- tokenhub/runtime/control.py +62 -0
- tokenhub/runtime/instance.py +79 -0
- tokenhub/runtime/manager.py +161 -0
- tokenhub/security/__init__.py +6 -0
- tokenhub/security/http.py +108 -0
- tokenhub/security/paths.py +204 -0
- tokenhub/security/redaction.py +13 -0
- tokenhub/security/windows_paths.py +223 -0
- tokenhub/server.py +29 -0
- tokenhub/settings.py +53 -0
- tokenhub/web/assets/index-B2a5VymX.css +1 -0
- tokenhub/web/assets/index-Ctyi0WkJ.js +40 -0
- tokenhub/web/favicon.svg +4 -0
- tokenhub/web/index.html +16 -0
- tokenhub-0.1.0.dist-info/METADATA +230 -0
- tokenhub-0.1.0.dist-info/RECORD +69 -0
- tokenhub-0.1.0.dist-info/WHEEL +5 -0
- tokenhub-0.1.0.dist-info/entry_points.txt +2 -0
- tokenhub-0.1.0.dist-info/licenses/LICENSE +21 -0
- tokenhub-0.1.0.dist-info/top_level.txt +1 -0
tokenhub/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""TokenHub package."""
|
tokenhub/__main__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Provider-neutral analytics over normalized local events."""
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Agent, model, and session views of the canonical observed workload."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from collections import defaultdict
|
|
5
|
+
from datetime import UTC
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from tokenhub.database.models import UsageEventRecord
|
|
9
|
+
|
|
10
|
+
_COUNTERS = (
|
|
11
|
+
"input_total_tokens", "output_total_tokens", "cache_read_tokens",
|
|
12
|
+
"cache_write_tokens", "reasoning_tokens",
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _session_key(event: UsageEventRecord) -> str:
|
|
17
|
+
# Stable across refreshes, with no original session id or filesystem path.
|
|
18
|
+
identity = f"{event.connector_id}\0{event.session_id or event.source_id}"
|
|
19
|
+
return hashlib.sha256(identity.encode()).hexdigest()[:24]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _summary(events: list[UsageEventRecord]) -> dict[str, Any]:
|
|
23
|
+
result: dict[str, Any] = {}
|
|
24
|
+
for field in _COUNTERS:
|
|
25
|
+
values = [getattr(event, field) for event in events if getattr(event, field) is not None]
|
|
26
|
+
result[field] = sum(values) if values else None
|
|
27
|
+
workloads = [
|
|
28
|
+
event.input_total_tokens + event.output_total_tokens
|
|
29
|
+
for event in events
|
|
30
|
+
if event.input_total_tokens is not None and event.output_total_tokens is not None
|
|
31
|
+
]
|
|
32
|
+
result.update(
|
|
33
|
+
workload_tokens=sum(workloads) if workloads else None,
|
|
34
|
+
event_count=len(events),
|
|
35
|
+
session_count=len({_session_key(event) for event in events}),
|
|
36
|
+
model_count=len({event.model_name for event in events if event.model_name is not None}),
|
|
37
|
+
incomplete_event_count=sum(
|
|
38
|
+
event.input_total_tokens is None or event.output_total_tokens is None for event in events
|
|
39
|
+
),
|
|
40
|
+
first_seen=min((event.timestamp for event in events), default=None),
|
|
41
|
+
last_seen=max((event.timestamp for event in events), default=None),
|
|
42
|
+
)
|
|
43
|
+
for field in ("first_seen", "last_seen"):
|
|
44
|
+
result[field] = result[field].replace(tzinfo=UTC).isoformat() if result[field] else None
|
|
45
|
+
return result
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def usage_breakdown(events: list[UsageEventRecord]) -> dict[str, Any]:
|
|
49
|
+
providers: dict[str, list[UsageEventRecord]] = defaultdict(list)
|
|
50
|
+
models: dict[tuple[str, str | None], list[UsageEventRecord]] = defaultdict(list)
|
|
51
|
+
sessions: dict[tuple[str, str], list[UsageEventRecord]] = defaultdict(list)
|
|
52
|
+
for event in events:
|
|
53
|
+
providers[event.provider].append(event)
|
|
54
|
+
models[event.provider, event.model_name].append(event)
|
|
55
|
+
sessions[event.provider, _session_key(event)].append(event)
|
|
56
|
+
model_rows = []
|
|
57
|
+
for (provider, model), group in models.items():
|
|
58
|
+
attributions = {event.model_attribution for event in group}
|
|
59
|
+
model_rows.append(dict(
|
|
60
|
+
_summary(group), provider=provider, model_name=model,
|
|
61
|
+
attribution=next(iter(attributions)) if len(attributions) == 1 else "mixed",
|
|
62
|
+
))
|
|
63
|
+
session_rows = []
|
|
64
|
+
for (provider, key), group in sessions.items():
|
|
65
|
+
session_models: dict[str | None, list[UsageEventRecord]] = defaultdict(list)
|
|
66
|
+
for event in group:
|
|
67
|
+
session_models[event.model_name].append(event)
|
|
68
|
+
session_rows.append(dict(
|
|
69
|
+
_summary(group), provider=provider, session_key=key,
|
|
70
|
+
models=[dict(_summary(rows), model_name=model) for model, rows in session_models.items()],
|
|
71
|
+
))
|
|
72
|
+
sort_key = lambda row: (-(row["workload_tokens"] or 0), row["provider"], row.get("model_name") or "")
|
|
73
|
+
return {
|
|
74
|
+
"totals": _summary(events),
|
|
75
|
+
"providers": sorted([dict(_summary(group), provider=provider) for provider, group in providers.items()], key=sort_key),
|
|
76
|
+
"models": sorted(model_rows, key=sort_key),
|
|
77
|
+
"sessions": sorted(session_rows, key=sort_key),
|
|
78
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Dashboard queries over observed, normalized usage only."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from tokenhub.analytics.breakdown import usage_breakdown
|
|
7
|
+
from tokenhub.database.repositories import UsageRepository
|
|
8
|
+
from tokenhub.domain.models import DashboardSummary
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class AnalyticsService:
|
|
12
|
+
def __init__(self, usage_repository: UsageRepository) -> None:
|
|
13
|
+
self.usage_repository = usage_repository
|
|
14
|
+
|
|
15
|
+
def dashboard(self) -> DashboardSummary:
|
|
16
|
+
return self.usage_repository.dashboard_totals()
|
|
17
|
+
|
|
18
|
+
def usage_breakdown(
|
|
19
|
+
self, start: datetime | None = None, end: datetime | None = None
|
|
20
|
+
) -> dict[str, Any]:
|
|
21
|
+
return usage_breakdown(self.usage_repository.observed_events(start, end))
|
tokenhub/api/__init__.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Composition root for the local TokenHub application.
|
|
2
|
+
|
|
3
|
+
The container owns startup and shutdown so that constructing the app has **no
|
|
4
|
+
side effects**: no database file, no migration, and no provider discovery until
|
|
5
|
+
the ASGI lifespan starts. It also owns the two pieces of state that must survive
|
|
6
|
+
across HTTP requests:
|
|
7
|
+
|
|
8
|
+
* the process-local discovery handoff (``SourceRepository`` pending candidates),
|
|
9
|
+
* the injected ``DiscoveryContext``, which tests replace after ``create_app``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
import shutil
|
|
17
|
+
import threading
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
from sqlalchemy import Engine
|
|
21
|
+
from sqlalchemy.orm import Session
|
|
22
|
+
|
|
23
|
+
from tokenhub.analytics.service import AnalyticsService
|
|
24
|
+
from tokenhub.connectors.protocol import DiscoveryContext
|
|
25
|
+
from tokenhub.connectors.registry import ConnectorRegistry
|
|
26
|
+
from tokenhub.database.migrations import run_migrations
|
|
27
|
+
from tokenhub.database.repositories import SourceRepository, UsageRepository
|
|
28
|
+
from tokenhub.database.session import create_engine_for
|
|
29
|
+
from tokenhub.discovery.service import DiscoveryService
|
|
30
|
+
from tokenhub.ingestion.collection import CollectionService
|
|
31
|
+
from tokenhub.ingestion.service import IngestionService
|
|
32
|
+
from tokenhub.settings import TokenHubSettings
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, slots=True)
|
|
36
|
+
class Services:
|
|
37
|
+
"""Long-lived collaborators shared by the HTTP routes."""
|
|
38
|
+
|
|
39
|
+
session: Session
|
|
40
|
+
source_repository: SourceRepository
|
|
41
|
+
usage_repository: UsageRepository
|
|
42
|
+
discovery: DiscoveryService
|
|
43
|
+
ingestion: IngestionService
|
|
44
|
+
analytics: AnalyticsService
|
|
45
|
+
collection: CollectionService
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class Container:
|
|
49
|
+
"""Startup, shutdown, and request serialization for one app instance."""
|
|
50
|
+
|
|
51
|
+
def __init__(self, settings: TokenHubSettings) -> None:
|
|
52
|
+
self.settings = settings
|
|
53
|
+
# One SQLite connection backs this single-user local server; the lock
|
|
54
|
+
# keeps sequential routes from interleaving on it.
|
|
55
|
+
self.lock = threading.RLock()
|
|
56
|
+
self._engine: Engine | None = None
|
|
57
|
+
self._services: Services | None = None
|
|
58
|
+
self._discovery_context: DiscoveryContext | None = None
|
|
59
|
+
self.collection_thread: threading.Thread | None = None
|
|
60
|
+
self._collection_stop = threading.Event()
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def services(self) -> Services:
|
|
64
|
+
"""Collaborators for the running app; raises before :meth:`start`."""
|
|
65
|
+
if self._services is None:
|
|
66
|
+
raise RuntimeError("container has not started")
|
|
67
|
+
return self._services
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def discovery_context(self) -> DiscoveryContext:
|
|
71
|
+
"""Discovery inputs, defaulting to the real home and process environment."""
|
|
72
|
+
if self._discovery_context is not None:
|
|
73
|
+
return self._discovery_context
|
|
74
|
+
return DiscoveryContext(
|
|
75
|
+
home=self.settings.home_directory,
|
|
76
|
+
environment=os.environ,
|
|
77
|
+
which=shutil.which,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
@discovery_context.setter
|
|
81
|
+
def discovery_context(self, context: DiscoveryContext) -> None:
|
|
82
|
+
"""Replace discovery inputs, including after startup."""
|
|
83
|
+
self._discovery_context = context
|
|
84
|
+
if self._services is not None:
|
|
85
|
+
self._services.discovery.context = context
|
|
86
|
+
|
|
87
|
+
def start(self) -> None:
|
|
88
|
+
"""Migrate the database, then build the services that routes depend on."""
|
|
89
|
+
run_migrations(self.settings)
|
|
90
|
+
engine = create_engine_for(self.settings)
|
|
91
|
+
session = Session(engine)
|
|
92
|
+
registry = ConnectorRegistry.default()
|
|
93
|
+
source_repository = SourceRepository(session)
|
|
94
|
+
source_repository.upgrade_codex_parser_versions()
|
|
95
|
+
usage_repository = UsageRepository(session)
|
|
96
|
+
self._engine = engine
|
|
97
|
+
discovery = DiscoveryService(registry, source_repository, self.discovery_context)
|
|
98
|
+
ingestion = IngestionService(source_repository, usage_repository, registry)
|
|
99
|
+
self._services = Services(
|
|
100
|
+
session=session,
|
|
101
|
+
source_repository=source_repository,
|
|
102
|
+
usage_repository=usage_repository,
|
|
103
|
+
discovery=discovery,
|
|
104
|
+
ingestion=ingestion,
|
|
105
|
+
analytics=AnalyticsService(usage_repository),
|
|
106
|
+
collection=CollectionService(source_repository, usage_repository, discovery, ingestion),
|
|
107
|
+
)
|
|
108
|
+
self._collect_safely()
|
|
109
|
+
self._collection_stop.clear()
|
|
110
|
+
self.collection_thread = threading.Thread(
|
|
111
|
+
target=self._collect_periodically, name="tokenhub-collection", daemon=True
|
|
112
|
+
)
|
|
113
|
+
self.collection_thread.start()
|
|
114
|
+
|
|
115
|
+
def collect(self) -> None:
|
|
116
|
+
with self.lock:
|
|
117
|
+
self.services.collection.run_once()
|
|
118
|
+
|
|
119
|
+
def _collect_periodically(self) -> None:
|
|
120
|
+
while not self._collection_stop.wait(self.settings.scan_interval_seconds):
|
|
121
|
+
self._collect_safely()
|
|
122
|
+
|
|
123
|
+
def _collect_safely(self) -> None:
|
|
124
|
+
try:
|
|
125
|
+
self.collect()
|
|
126
|
+
except Exception: # noqa: BLE001 - startup and later collection must remain available
|
|
127
|
+
# Do not log exception text, which may contain a provider path.
|
|
128
|
+
logging.getLogger(__name__).error("Automatic collection failed; retrying on the next scan")
|
|
129
|
+
with self.lock:
|
|
130
|
+
self.services.session.rollback()
|
|
131
|
+
self.services.collection.failed_source_count = max(
|
|
132
|
+
1, self.services.collection.failed_source_count
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
def stop(self) -> None:
|
|
136
|
+
"""Release the session and engine; the database file stays behind."""
|
|
137
|
+
self._collection_stop.set()
|
|
138
|
+
if self.collection_thread is not None:
|
|
139
|
+
self.collection_thread.join()
|
|
140
|
+
self.collection_thread = None
|
|
141
|
+
if self._services is not None:
|
|
142
|
+
self._services.session.close()
|
|
143
|
+
self._services = None
|
|
144
|
+
if self._engine is not None:
|
|
145
|
+
self._engine.dispose()
|
|
146
|
+
self._engine = None
|
tokenhub/api/routes.py
ADDED
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"""Thin, versioned HTTP routes over the TokenHub service layer.
|
|
2
|
+
|
|
3
|
+
Routes own transport concerns only — status codes, JSON projection, and error
|
|
4
|
+
mapping. All persistence, discovery, parsing, and aggregation stay in the
|
|
5
|
+
services so the local API cannot drift from the tested behavior.
|
|
6
|
+
|
|
7
|
+
Every projection here is deliberately path-free and record-free: the HTTP
|
|
8
|
+
surface exposes identifiers, states, counts, and timestamps, never a canonical
|
|
9
|
+
path, approved root, provider path, credential, prompt, or raw record.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
from datetime import datetime
|
|
16
|
+
from typing import Annotated
|
|
17
|
+
|
|
18
|
+
from fastapi import APIRouter, HTTPException, Query, Request
|
|
19
|
+
|
|
20
|
+
from tokenhub.api.container import Container
|
|
21
|
+
|
|
22
|
+
# The service layer's own return type for an approval. The HTTP layer only
|
|
23
|
+
# reads it — it never queries the ORM itself — so this is a data dependency on
|
|
24
|
+
# the service contract, not a persistence leak into transport code.
|
|
25
|
+
from tokenhub.database.models import SourceRecord
|
|
26
|
+
from tokenhub.domain.models import (
|
|
27
|
+
DashboardSummary,
|
|
28
|
+
ImportOutcome,
|
|
29
|
+
RebuildOutcome,
|
|
30
|
+
)
|
|
31
|
+
from tokenhub.ingestion.service import (
|
|
32
|
+
SourceNotApprovedError,
|
|
33
|
+
SourceNotFoundError,
|
|
34
|
+
UnsupportedSourceError,
|
|
35
|
+
)
|
|
36
|
+
from tokenhub.security.redaction import redact_sensitive
|
|
37
|
+
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
router = APIRouter(prefix="/api/v1")
|
|
41
|
+
|
|
42
|
+
_ERROR_MAP: tuple[tuple[type[Exception], int, str], ...] = (
|
|
43
|
+
(SourceNotFoundError, 404, "Unknown source"),
|
|
44
|
+
(SourceNotApprovedError, 409, "Source is not approved"),
|
|
45
|
+
(UnsupportedSourceError, 422, "Source is not supported"),
|
|
46
|
+
(ValueError, 400, "Invalid source path"),
|
|
47
|
+
(OSError, 400, "Source is unavailable"),
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def container_for(request: Request) -> Container:
|
|
52
|
+
"""The app's container, injected by the factory."""
|
|
53
|
+
return request.app.state.container
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _service_failure(error: Exception) -> HTTPException:
|
|
57
|
+
"""Map a service exception to a status code and a path-free message.
|
|
58
|
+
|
|
59
|
+
Known service failures get a fixed, path-free detail. Anything else becomes
|
|
60
|
+
a generic internal error: a route must never hand a provider path, a raw
|
|
61
|
+
record, or a credential to the client, and the log line goes through the
|
|
62
|
+
central redactor.
|
|
63
|
+
"""
|
|
64
|
+
for error_type, status, detail in _ERROR_MAP:
|
|
65
|
+
if isinstance(error, error_type):
|
|
66
|
+
return HTTPException(status_code=status, detail=redact_sensitive(detail))
|
|
67
|
+
logger.error("unhandled TokenHub service failure: %s", redact_sensitive(str(error)))
|
|
68
|
+
return HTTPException(status_code=500, detail="Internal error")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _freshness_payload(summary: DashboardSummary) -> list[dict[str, object]]:
|
|
72
|
+
return [
|
|
73
|
+
{
|
|
74
|
+
"source_id": item.source_id,
|
|
75
|
+
"state": item.state.value,
|
|
76
|
+
"latest_event_at": (
|
|
77
|
+
item.latest_event_at.isoformat()
|
|
78
|
+
if item.latest_event_at is not None
|
|
79
|
+
else None
|
|
80
|
+
),
|
|
81
|
+
"unsupported_records": item.unsupported_records,
|
|
82
|
+
}
|
|
83
|
+
for item in summary.source_freshness
|
|
84
|
+
]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _quality_payload(summary: DashboardSummary) -> dict[str, int]:
|
|
88
|
+
return {quality.value: count for quality, count in summary.quality_counts.items()}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _summary_payload(summary: DashboardSummary) -> dict[str, object]:
|
|
92
|
+
return {
|
|
93
|
+
"workload_tokens": summary.workload_tokens,
|
|
94
|
+
"input_total_tokens": summary.input_total_tokens,
|
|
95
|
+
"output_total_tokens": summary.output_total_tokens,
|
|
96
|
+
"cache_read_tokens": summary.cache_read_tokens,
|
|
97
|
+
"cache_write_tokens": summary.cache_write_tokens,
|
|
98
|
+
"reasoning_tokens": summary.reasoning_tokens,
|
|
99
|
+
"event_count": summary.event_count,
|
|
100
|
+
"quality_counts": _quality_payload(summary),
|
|
101
|
+
"source_freshness": _freshness_payload(summary),
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _import_payload(outcome: ImportOutcome) -> dict[str, object]:
|
|
106
|
+
return {
|
|
107
|
+
"inserted_events": outcome.inserted_events,
|
|
108
|
+
"duplicate_events": outcome.duplicate_events,
|
|
109
|
+
"partial_final_record": outcome.partial_final_record,
|
|
110
|
+
"unsupported_records": outcome.unsupported_records,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _approval_payload(source: SourceRecord) -> dict[str, object]:
|
|
115
|
+
return {
|
|
116
|
+
"source_id": source.source_id,
|
|
117
|
+
"provider": source.provider,
|
|
118
|
+
"display_name": source.display_name,
|
|
119
|
+
"state": source.state,
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@router.get("/status")
|
|
124
|
+
def status() -> dict[str, str]:
|
|
125
|
+
"""Liveness for the local UI and for startup checks."""
|
|
126
|
+
return {"status": "ok"}
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
@router.get("/discovery")
|
|
130
|
+
def discovery(request: Request) -> dict[str, object]:
|
|
131
|
+
"""Report detected providers. Presence only — no source file is read."""
|
|
132
|
+
container = container_for(request)
|
|
133
|
+
with container.lock:
|
|
134
|
+
results = container.services.discovery.discover()
|
|
135
|
+
return {
|
|
136
|
+
"providers": [
|
|
137
|
+
{
|
|
138
|
+
"connector_id": result.connector_id,
|
|
139
|
+
"display_name": result.display_name,
|
|
140
|
+
"provider": result.provider.value if result.provider is not None else None,
|
|
141
|
+
"state": result.state.value,
|
|
142
|
+
"confidence": result.confidence.value,
|
|
143
|
+
"evidence_codes": list(result.evidence_codes),
|
|
144
|
+
"sources": [source.model_dump(mode="json") for source in result.sources],
|
|
145
|
+
}
|
|
146
|
+
for result in results
|
|
147
|
+
]
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@router.get("/collection")
|
|
152
|
+
def collection_status(request: Request) -> dict[str, object]:
|
|
153
|
+
container = container_for(request)
|
|
154
|
+
with container.lock:
|
|
155
|
+
return container.services.collection.status(container.settings.scan_interval_seconds)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _collection_connector(provider: str) -> str:
|
|
159
|
+
connectors = {
|
|
160
|
+
"codex": "codex-local",
|
|
161
|
+
"claude_code": "claude-code-local",
|
|
162
|
+
"hermes": "hermes-local",
|
|
163
|
+
"vscode_copilot": "vscode-copilot-local",
|
|
164
|
+
"antigravity": "antigravity-local",
|
|
165
|
+
}
|
|
166
|
+
if provider not in connectors:
|
|
167
|
+
raise HTTPException(status_code=422, detail="Source is not supported")
|
|
168
|
+
return connectors[provider]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
@router.post("/collection/{provider}/enable")
|
|
172
|
+
def enable_provider_collection(provider: str, request: Request) -> dict[str, object]:
|
|
173
|
+
"""Explicitly include existing and future sources under the discovered root."""
|
|
174
|
+
container = container_for(request)
|
|
175
|
+
connector_id = _collection_connector(provider)
|
|
176
|
+
with container.lock:
|
|
177
|
+
try:
|
|
178
|
+
container.services.collection.enable(connector_id)
|
|
179
|
+
except Exception as error:
|
|
180
|
+
raise _service_failure(error) from error
|
|
181
|
+
return container.services.collection.status(container.settings.scan_interval_seconds)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
@router.post("/collection/{provider}/disable")
|
|
185
|
+
def disable_provider_collection(provider: str, request: Request) -> dict[str, object]:
|
|
186
|
+
"""Stop automatically approving new sources; existing approvals remain."""
|
|
187
|
+
container = container_for(request)
|
|
188
|
+
connector_id = _collection_connector(provider)
|
|
189
|
+
with container.lock:
|
|
190
|
+
container.services.source_repository.disable_auto_import(connector_id)
|
|
191
|
+
return container.services.collection.status(container.settings.scan_interval_seconds)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
@router.post("/sources/{source_id}/approve")
|
|
195
|
+
def approve_source(source_id: str, request: Request) -> dict[str, object]:
|
|
196
|
+
"""Approve reading one discovered source; requires a same-origin request."""
|
|
197
|
+
container = container_for(request)
|
|
198
|
+
with container.lock:
|
|
199
|
+
try:
|
|
200
|
+
source = container.services.ingestion.approve(source_id)
|
|
201
|
+
except Exception as error: # mapped to a safe status below
|
|
202
|
+
raise _service_failure(error) from error
|
|
203
|
+
return _approval_payload(source)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
@router.post("/sources/{source_id}/rescan")
|
|
207
|
+
def rescan_source(source_id: str, request: Request) -> dict[str, object]:
|
|
208
|
+
"""Import new records from an approved source."""
|
|
209
|
+
container = container_for(request)
|
|
210
|
+
with container.lock:
|
|
211
|
+
try:
|
|
212
|
+
outcome = container.services.ingestion.rescan(source_id)
|
|
213
|
+
except Exception as error: # mapped to a safe status below
|
|
214
|
+
raise _service_failure(error) from error
|
|
215
|
+
return _import_payload(outcome)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
@router.post("/rebuild")
|
|
219
|
+
def rebuild(request: Request) -> dict[str, object]:
|
|
220
|
+
"""Re-derive normalized usage for every approved source."""
|
|
221
|
+
container = container_for(request)
|
|
222
|
+
with container.lock:
|
|
223
|
+
try:
|
|
224
|
+
outcome: RebuildOutcome = container.services.ingestion.rebuild()
|
|
225
|
+
except Exception as error: # mapped to a safe status below
|
|
226
|
+
raise _service_failure(error) from error
|
|
227
|
+
return {
|
|
228
|
+
"inserted_events": outcome.inserted_events,
|
|
229
|
+
"failed_source_ids": list(outcome.failed_source_ids),
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
@router.get("/dashboard")
|
|
234
|
+
def dashboard(request: Request) -> dict[str, object]:
|
|
235
|
+
"""Observed delta totals; unknown metrics stay null rather than becoming zero."""
|
|
236
|
+
container = container_for(request)
|
|
237
|
+
with container.lock:
|
|
238
|
+
summary = container.services.analytics.dashboard()
|
|
239
|
+
return _summary_payload(summary)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
@router.get("/data-quality")
|
|
243
|
+
def data_quality(request: Request) -> dict[str, object]:
|
|
244
|
+
"""The same quality and freshness view the dashboard shows."""
|
|
245
|
+
container = container_for(request)
|
|
246
|
+
with container.lock:
|
|
247
|
+
summary = container.services.analytics.dashboard()
|
|
248
|
+
return {
|
|
249
|
+
"quality_counts": _quality_payload(summary),
|
|
250
|
+
"source_freshness": _freshness_payload(summary),
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
@router.get("/usage")
|
|
255
|
+
def usage(
|
|
256
|
+
request: Request,
|
|
257
|
+
start: Annotated[datetime | None, Query(alias="from")] = None,
|
|
258
|
+
end: Annotated[datetime | None, Query(alias="to")] = None,
|
|
259
|
+
) -> dict[str, object]:
|
|
260
|
+
"""Path-free usage grouped by agent, recorded model, and opaque session."""
|
|
261
|
+
if any(value is not None and value.utcoffset() is None for value in (start, end)):
|
|
262
|
+
raise HTTPException(status_code=422, detail="Date bounds require a timezone")
|
|
263
|
+
if start is not None and end is not None and start >= end:
|
|
264
|
+
raise HTTPException(status_code=422, detail="Date range must end after it begins")
|
|
265
|
+
container = container_for(request)
|
|
266
|
+
with container.lock:
|
|
267
|
+
return container.services.analytics.usage_breakdown(start, end)
|
tokenhub/app.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""FastAPI application factory for the local TokenHub server.
|
|
2
|
+
|
|
3
|
+
Two guards run before every route, in this order:
|
|
4
|
+
|
|
5
|
+
1. **Host** — only loopback host forms with a canonical port are accepted, so a
|
|
6
|
+
DNS-rebinding page cannot reach the API. Applies to the static UI as well.
|
|
7
|
+
2. **Origin** — state-changing routes must carry the request's own loopback
|
|
8
|
+
origin; a missing, delegated, or duplicated Origin is refused before any
|
|
9
|
+
service sees the request.
|
|
10
|
+
|
|
11
|
+
Neither guard adds CORS headers, and forwarded-host/proto headers are never
|
|
12
|
+
consulted: the server is not behind a proxy by design.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections.abc import AsyncIterator
|
|
18
|
+
from contextlib import asynccontextmanager
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from fastapi import FastAPI, Request, Response
|
|
22
|
+
from fastapi.responses import JSONResponse
|
|
23
|
+
from fastapi.staticfiles import StaticFiles
|
|
24
|
+
from starlette.middleware.base import RequestResponseEndpoint
|
|
25
|
+
|
|
26
|
+
from tokenhub.api.container import Container
|
|
27
|
+
from tokenhub.api.routes import router
|
|
28
|
+
from tokenhub.runtime import RuntimeControl, runtime_router
|
|
29
|
+
from tokenhub.security.http import is_loopback_host, is_same_origin
|
|
30
|
+
from tokenhub.settings import TokenHubSettings
|
|
31
|
+
|
|
32
|
+
#: The built UI is package data so installed wheels and source builds serve it.
|
|
33
|
+
FRONTEND_DIST = Path(__file__).resolve().parent / "web"
|
|
34
|
+
|
|
35
|
+
_READ_ONLY_METHODS = frozenset({"GET", "HEAD"})
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def create_app(
|
|
39
|
+
settings: TokenHubSettings | None = None,
|
|
40
|
+
runtime_control: RuntimeControl | None = None,
|
|
41
|
+
) -> FastAPI:
|
|
42
|
+
"""Build the app. No database, migration, or discovery work happens here."""
|
|
43
|
+
container = Container(settings if settings is not None else TokenHubSettings())
|
|
44
|
+
|
|
45
|
+
@asynccontextmanager
|
|
46
|
+
async def lifespan(_: FastAPI) -> AsyncIterator[None]:
|
|
47
|
+
container.start()
|
|
48
|
+
try:
|
|
49
|
+
yield
|
|
50
|
+
finally:
|
|
51
|
+
container.stop()
|
|
52
|
+
|
|
53
|
+
app = FastAPI(title="TokenHub", lifespan=lifespan)
|
|
54
|
+
app.state.container = container
|
|
55
|
+
|
|
56
|
+
@app.middleware("http")
|
|
57
|
+
async def local_request_guard(
|
|
58
|
+
request: Request, call_next: RequestResponseEndpoint
|
|
59
|
+
) -> Response:
|
|
60
|
+
port = container.settings.port
|
|
61
|
+
hosts = request.headers.getlist("host")
|
|
62
|
+
if len(hosts) != 1 or not is_loopback_host(hosts[0], port):
|
|
63
|
+
return JSONResponse(status_code=400, content={"detail": "Invalid Host header"})
|
|
64
|
+
if request.method not in _READ_ONLY_METHODS:
|
|
65
|
+
origins = request.headers.getlist("origin")
|
|
66
|
+
if len(origins) != 1 or not is_same_origin(origins[0], hosts[0], port):
|
|
67
|
+
return JSONResponse(
|
|
68
|
+
status_code=403, content={"detail": "Invalid Origin header"}
|
|
69
|
+
)
|
|
70
|
+
return await call_next(request)
|
|
71
|
+
|
|
72
|
+
app.include_router(router)
|
|
73
|
+
if runtime_control is not None:
|
|
74
|
+
app.include_router(runtime_router(runtime_control))
|
|
75
|
+
if FRONTEND_DIST.is_dir():
|
|
76
|
+
app.mount("/", StaticFiles(directory=FRONTEND_DIST, html=True), name="frontend")
|
|
77
|
+
return app
|
tokenhub/cli.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Commands for the local TokenHub dashboard and its background server."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
import webbrowser
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
|
|
11
|
+
from tokenhub.runtime.manager import RuntimeManager
|
|
12
|
+
from tokenhub.server import run_server
|
|
13
|
+
from tokenhub.settings import TokenHubSettings
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def loopback_url(settings: TokenHubSettings) -> str:
|
|
17
|
+
"""The only URL this server is reachable at."""
|
|
18
|
+
host = "[::1]" if settings.host == "::1" else settings.host
|
|
19
|
+
return f"http://{host}:{settings.port}/"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
23
|
+
"""Start or manage the local dashboard; start opens it in a browser."""
|
|
24
|
+
args = list(sys.argv[1:] if argv is None else argv)
|
|
25
|
+
if args and args[0] == "_serve":
|
|
26
|
+
if len(args) != 2:
|
|
27
|
+
raise ValueError("internal server requires one port")
|
|
28
|
+
token = os.environ.pop("TOKENHUB_INTERNAL_CONTROL_TOKEN")
|
|
29
|
+
run_server(TokenHubSettings(port=int(args[1])), token)
|
|
30
|
+
return 0
|
|
31
|
+
|
|
32
|
+
parser = argparse.ArgumentParser(prog="tokenhub")
|
|
33
|
+
parser.add_argument(
|
|
34
|
+
"command", nargs="?", choices=("start", "status", "open", "stop"), default="start"
|
|
35
|
+
)
|
|
36
|
+
parser.add_argument("--port", type=int, help="loopback port for a new server")
|
|
37
|
+
parser.add_argument("--no-open", action="store_true", help="do not open a browser after start")
|
|
38
|
+
options = parser.parse_args(args)
|
|
39
|
+
manager = RuntimeManager(TokenHubSettings().data_directory)
|
|
40
|
+
try:
|
|
41
|
+
if options.command == "stop":
|
|
42
|
+
print("Token Hub stopped" if manager.stop() else "Token Hub is not running")
|
|
43
|
+
return 0
|
|
44
|
+
if options.command == "status":
|
|
45
|
+
url = manager.status()
|
|
46
|
+
print(url if url else "Token Hub is not running")
|
|
47
|
+
return 0 if url else 1
|
|
48
|
+
if options.command == "open":
|
|
49
|
+
url = manager.status()
|
|
50
|
+
if url is None:
|
|
51
|
+
print("Token Hub is not running; run tokenhub start")
|
|
52
|
+
return 1
|
|
53
|
+
else:
|
|
54
|
+
url = manager.start(options.port)
|
|
55
|
+
except (OSError, RuntimeError, ValueError) as error:
|
|
56
|
+
print(f"Token Hub: {error}", file=sys.stderr)
|
|
57
|
+
return 1
|
|
58
|
+
|
|
59
|
+
print(url)
|
|
60
|
+
if options.command == "open" or not options.no_open:
|
|
61
|
+
try:
|
|
62
|
+
webbrowser.open(url)
|
|
63
|
+
except Exception as error: # noqa: BLE001 - browser implementations vary by platform
|
|
64
|
+
print(f"Could not open browser: {error}. Open {url}", file=sys.stderr)
|
|
65
|
+
return 0
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Provider connectors for safe local discovery and approved scans."""
|