token-runtime 0.1.0a2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,111 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from typing import TYPE_CHECKING, Literal
5
+
6
+ if TYPE_CHECKING:
7
+ from .compatibility import CompatibilityRecord, CompatibilityRegistry
8
+
9
+
10
+ UnknownBool = bool | Literal["unknown"]
11
+
12
+
13
+ @dataclass(frozen=True, order=True, slots=True)
14
+ class CapabilityKey:
15
+ client_family: str
16
+ protocol_family: str
17
+ provider_family: str
18
+ model_family: str | None = None
19
+ client_version: str = ""
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class CapabilityProfile:
24
+ key: CapabilityKey
25
+ context_window_semantics: str = "unknown"
26
+ tokenizer_family: str = "unknown"
27
+ prompt_cache: str = "unknown"
28
+ server_cache: str = "unknown"
29
+ stateful_sessions: UnknownBool = "unknown"
30
+ native_context_management: UnknownBool = "unknown"
31
+ structured_output: UnknownBool = "unknown"
32
+ tool_calls: UnknownBool = "unknown"
33
+ multimodal: UnknownBool = "unknown"
34
+ reasoning_state: str = "unknown"
35
+ streaming: UnknownBool = "unknown"
36
+ exact_byte_preservation: UnknownBool = "unknown"
37
+ evidence_id: str = "unknown"
38
+
39
+ def to_primitive(self) -> dict[str, object]:
40
+ key: dict[str, object] = {
41
+ "client_family": self.key.client_family,
42
+ "protocol_family": self.key.protocol_family,
43
+ "provider_family": self.key.provider_family,
44
+ "model_family": self.key.model_family,
45
+ }
46
+ if self.key.client_version:
47
+ key["client_version"] = self.key.client_version
48
+ return {
49
+ "key": key,
50
+ "context_window_semantics": self.context_window_semantics,
51
+ "tokenizer_family": self.tokenizer_family,
52
+ "prompt_cache": self.prompt_cache,
53
+ "server_cache": self.server_cache,
54
+ "stateful_sessions": self.stateful_sessions,
55
+ "native_context_management": self.native_context_management,
56
+ "structured_output": self.structured_output,
57
+ "tool_calls": self.tool_calls,
58
+ "multimodal": self.multimodal,
59
+ "reasoning_state": self.reasoning_state,
60
+ "streaming": self.streaming,
61
+ "exact_byte_preservation": self.exact_byte_preservation,
62
+ "evidence_id": self.evidence_id,
63
+ }
64
+
65
+
66
+ class DuplicateCapabilityError(ValueError):
67
+ pass
68
+
69
+
70
+ class CapabilityRegistry:
71
+ def __init__(self) -> None:
72
+ self._profiles: dict[CapabilityKey, CapabilityProfile] = {}
73
+
74
+ def register(self, profile: CapabilityProfile) -> CapabilityProfile:
75
+ existing = self._profiles.get(profile.key)
76
+ if existing is None:
77
+ self._profiles[profile.key] = profile
78
+ return profile
79
+ if existing == profile:
80
+ return existing
81
+ raise DuplicateCapabilityError(f"conflicting capability profile: {profile.key!r}")
82
+
83
+ def get(self, key: CapabilityKey) -> CapabilityProfile | None:
84
+ return self._profiles.get(key)
85
+
86
+ def snapshot(self) -> tuple[CapabilityProfile, ...]:
87
+ return tuple(self._profiles[key] for key in sorted(self._profiles))
88
+
89
+
90
+ @dataclass(frozen=True, slots=True)
91
+ class CapabilityDetection:
92
+ key: CapabilityKey
93
+ profile: CapabilityProfile | None
94
+ compatibility: "CompatibilityRecord"
95
+
96
+
97
+ class CapabilityDetector:
98
+ def __init__(
99
+ self,
100
+ capabilities: CapabilityRegistry,
101
+ compatibility: "CompatibilityRegistry",
102
+ ) -> None:
103
+ self.capabilities = capabilities
104
+ self.compatibility = compatibility
105
+
106
+ def detect(self, key: CapabilityKey) -> CapabilityDetection:
107
+ return CapabilityDetection(
108
+ key=key,
109
+ profile=self.capabilities.get(key),
110
+ compatibility=self.compatibility.get(key),
111
+ )
token_runtime/cli.py ADDED
@@ -0,0 +1,284 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import io
5
+ import json
6
+ from pathlib import Path
7
+ import socket
8
+ import sys
9
+ import tempfile
10
+ from urllib.parse import urlparse
11
+
12
+ from .benchmark import BenchmarkCase, benchmark_cases
13
+ from .config import (
14
+ TokenConfig,
15
+ default_config_path,
16
+ default_state_dir,
17
+ load_config,
18
+ save_config,
19
+ )
20
+ from .engine import OptimizationEngine
21
+ from .gateway import GatewayCore, serve
22
+ from .integrations import (
23
+ apply_plan,
24
+ detect_integrations,
25
+ plan_install,
26
+ uninstall_integration,
27
+ )
28
+ from .metrics import MetricsStore
29
+ from .model import ContextBlock, RequestEnvelope
30
+ from .planner import ContextPlanner
31
+ from .store import RecoveryStore
32
+ from .terminal_ui import render_welcome
33
+
34
+
35
+ def _add_config_argument(parser: argparse.ArgumentParser) -> None:
36
+ parser.add_argument("--config", default=None)
37
+
38
+
39
+ def build_parser() -> argparse.ArgumentParser:
40
+ parser = argparse.ArgumentParser(prog="token")
41
+ subs = parser.add_subparsers(dest="command")
42
+ for name in ("serve", "status", "doctor"):
43
+ sub = subs.add_parser(name)
44
+ _add_config_argument(sub)
45
+
46
+ optimize = subs.add_parser("optimize")
47
+ _add_config_argument(optimize)
48
+ optimize.add_argument("--endpoint", default="/v1/responses")
49
+
50
+ benchmark = subs.add_parser("benchmark")
51
+ _add_config_argument(benchmark)
52
+ benchmark.add_argument("--file", default=None)
53
+
54
+ install = subs.add_parser("install")
55
+ _add_config_argument(install)
56
+ install.add_argument("--root", default=None)
57
+ install.add_argument("--upstream", default=None)
58
+ install.add_argument("--dry-run", action="store_true")
59
+
60
+ uninstall = subs.add_parser("uninstall")
61
+ _add_config_argument(uninstall)
62
+ uninstall.add_argument("--root", default=None)
63
+ uninstall.add_argument("--dry-run", action="store_true")
64
+
65
+ return parser
66
+
67
+
68
+ def _config_path(value: str | None) -> Path:
69
+ return Path(value) if value else default_config_path()
70
+
71
+
72
+ def _runtime(config):
73
+ state = Path(config.state_dir)
74
+ state.mkdir(parents=True, exist_ok=True)
75
+ metrics = MetricsStore(state / "metrics.db")
76
+ engine = OptimizationEngine(
77
+ planner=ContextPlanner(),
78
+ store=RecoveryStore(state / "recovery.db"),
79
+ )
80
+ return GatewayCore(engine=engine, metrics=metrics), metrics
81
+
82
+
83
+ def _read_input_bytes(stdin) -> bytes:
84
+ if hasattr(stdin, "buffer"):
85
+ return stdin.buffer.read()
86
+ value = stdin.read()
87
+ return value.encode("utf-8") if isinstance(value, str) else value
88
+
89
+ def _reachable(host: str, port: int, timeout: float = 0.5) -> bool:
90
+ try:
91
+ with socket.create_connection((host, port), timeout=timeout):
92
+ return True
93
+ except OSError:
94
+ return False
95
+
96
+
97
+ def doctor_report(config_path: Path) -> dict[str, bool]:
98
+ report = {
99
+ "config_ok": False,
100
+ "state_ok": False,
101
+ "gateway_reachable": False,
102
+ "upstream_reachable": False,
103
+ }
104
+ config = load_config(config_path)
105
+ report["config_ok"] = True
106
+
107
+ state = Path(config.state_dir)
108
+ try:
109
+ state.mkdir(parents=True, exist_ok=True)
110
+ with tempfile.NamedTemporaryFile(dir=state):
111
+ pass
112
+ RecoveryStore(state / "recovery.db")
113
+ MetricsStore(state / "metrics.db")
114
+ report["state_ok"] = True
115
+ except OSError:
116
+ return report
117
+
118
+ report["gateway_reachable"] = _reachable(config.host, config.port)
119
+ parsed = urlparse(config.upstream)
120
+ upstream_port = parsed.port or (443 if parsed.scheme == "https" else 80)
121
+ report["upstream_reachable"] = bool(parsed.hostname) and _reachable(
122
+ parsed.hostname, upstream_port
123
+ )
124
+ return report
125
+
126
+ def _write_json(stdout, value) -> None:
127
+ stdout.write(json.dumps(value, sort_keys=True) + "\n")
128
+
129
+
130
+ def _root_path(value: str | None) -> Path:
131
+ return Path(value).expanduser() if value else Path.home()
132
+
133
+
134
+ def _install_config(path: Path, args, root: Path) -> TokenConfig:
135
+ if path.exists():
136
+ return load_config(path)
137
+ if not args.upstream:
138
+ raise FileNotFoundError("TOKEN config is missing; pass --upstream to bootstrap it")
139
+ config = TokenConfig(
140
+ upstream=args.upstream,
141
+ host="127.0.0.1",
142
+ port=8788,
143
+ state_dir=str(default_state_dir(root)),
144
+ )
145
+ if not args.dry_run:
146
+ save_config(config, path)
147
+ return config
148
+
149
+
150
+ def _load_benchmark_cases(path: str | Path) -> list[BenchmarkCase]:
151
+ data = json.loads(Path(path).read_text(encoding="utf-8"))
152
+ if not isinstance(data, dict) or not isinstance(data.get("cases"), list):
153
+ raise ValueError("benchmark file must contain a cases array")
154
+ cases: list[BenchmarkCase] = []
155
+ for index, item in enumerate(data["cases"]):
156
+ if not isinstance(item, dict) or not isinstance(item.get("blocks"), list):
157
+ raise ValueError(f"benchmark case {index} is invalid")
158
+ blocks = []
159
+ for block_index, raw in enumerate(item["blocks"]):
160
+ if not isinstance(raw, dict):
161
+ raise ValueError(f"benchmark block {index}:{block_index} is invalid")
162
+ blocks.append(ContextBlock(
163
+ id=str(raw.get("id", f"b{block_index}")),
164
+ kind=str(raw["kind"]),
165
+ text=str(raw["text"]),
166
+ role=raw.get("role"),
167
+ turn_index=int(raw.get("turn_index", 0)),
168
+ ))
169
+ cases.append(BenchmarkCase(str(item.get("name", f"case-{index}")), RequestEnvelope(tuple(blocks))))
170
+ return cases
171
+
172
+
173
+ def main(argv=None, *, stdin=None, stdout=None) -> int:
174
+ args = build_parser().parse_args(argv)
175
+ stdin = sys.stdin if stdin is None else stdin
176
+ stdout = sys.stdout if stdout is None else stdout
177
+ if args.command is None:
178
+ render_welcome(stdout)
179
+ return 0
180
+ path = _config_path(args.config)
181
+
182
+ if args.command == "install":
183
+ root = _root_path(args.root)
184
+ config = _install_config(path, args, root)
185
+ gateway_url = f"http://{config.host}:{config.port}/v1"
186
+ integrations = detect_integrations(root)
187
+ rows = []
188
+ active = sidecars = blocked = 0
189
+ for integration in integrations:
190
+ plan = plan_install(integration, gateway_url=gateway_url)
191
+ result = apply_plan(plan, apply=not args.dry_run)
192
+ activation = "managed_gateway" if plan.mode in {"managed_block", "managed_toml"} else "detected_sidecar_only"
193
+ if not plan.safe_to_apply:
194
+ blocked += 1
195
+ elif not args.dry_run and activation == "managed_gateway":
196
+ active += 1
197
+ elif not args.dry_run:
198
+ sidecars += 1
199
+ rows.append({
200
+ "kind": integration.kind,
201
+ "activation": activation,
202
+ "changed": result.changed,
203
+ "reason": result.reason,
204
+ "target": str(result.target_path),
205
+ })
206
+ _write_json(stdout, {
207
+ "mode": "dry_run" if args.dry_run else "apply",
208
+ "detected": len(integrations),
209
+ "applied": active,
210
+ "sidecars": sidecars,
211
+ "blocked": blocked,
212
+ "integrations": rows,
213
+ })
214
+ return 1 if blocked else 0
215
+
216
+ if args.command == "uninstall":
217
+ root = _root_path(args.root)
218
+ integrations = detect_integrations(root)
219
+ rows = []
220
+ removed = 0
221
+ for integration in integrations:
222
+ result = uninstall_integration(integration, apply=not args.dry_run)
223
+ if not args.dry_run and result.changed:
224
+ removed += 1
225
+ rows.append({
226
+ "kind": integration.kind,
227
+ "changed": result.changed,
228
+ "reason": result.reason,
229
+ "target": str(result.target_path),
230
+ })
231
+ _write_json(stdout, {
232
+ "mode": "dry_run" if args.dry_run else "apply",
233
+ "detected": len(integrations),
234
+ "removed": removed,
235
+ "integrations": rows,
236
+ })
237
+ return 0
238
+
239
+ config = load_config(path)
240
+ if args.command == "serve":
241
+ core, _ = _runtime(config)
242
+ serve(config.host, config.port, upstream=config.upstream, core=core)
243
+ return 0
244
+ if args.command == "status":
245
+ _, metrics = _runtime(config)
246
+ _write_json(stdout, metrics.summary())
247
+ return 0
248
+ if args.command == "doctor":
249
+ report = doctor_report(path)
250
+ _write_json(stdout, report)
251
+ return 0 if all(report.values()) else 1
252
+ if args.command == "optimize":
253
+ core, _ = _runtime(config)
254
+ prepared = core.prepare(args.endpoint, _read_input_bytes(stdin))
255
+ stdout.write(prepared.body.decode("utf-8", errors="surrogateescape"))
256
+ return 0
257
+ if args.command == "benchmark":
258
+ if not args.file:
259
+ raise ValueError("benchmark requires --file")
260
+ core, _ = _runtime(config)
261
+ report = benchmark_cases(_load_benchmark_cases(args.file), core.engine)
262
+ all_fidelity = all(row.protected_byte_fidelity == 1.0 for row in report.cases)
263
+ all_recoverable = all(row.recoverable for row in report.cases)
264
+ _write_json(stdout, {
265
+ "baseline_estimated_tokens": report.baseline_estimated_tokens,
266
+ "optimized_estimated_tokens": report.optimized_estimated_tokens,
267
+ "total_saved_pct": report.total_saved_pct,
268
+ "eligible_saved_pct": report.eligible_saved_pct,
269
+ "optimized_count": report.optimized_count,
270
+ "bypass_count": report.bypass_count,
271
+ "all_protected_fidelity": all_fidelity,
272
+ "all_recoverable": all_recoverable,
273
+ "offline_gate": bool(report.eligible_saved_pct >= 20 and all_fidelity and all_recoverable),
274
+ "cases": [{
275
+ "name": row.name,
276
+ "decision": row.decision,
277
+ "saved_pct": row.saved_pct,
278
+ "reasons": row.reasons,
279
+ "protected_byte_fidelity": row.protected_byte_fidelity,
280
+ "recoverable": row.recoverable,
281
+ } for row in report.cases],
282
+ })
283
+ return 0
284
+ return 2
@@ -0,0 +1,115 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+
5
+ from .capabilities import CapabilityDetector, CapabilityKey, CapabilityProfile, CapabilityRegistry
6
+ from .compatibility import CompatibilityRecord, CompatibilityRegistry, CompatibilityState
7
+ from .openai_certification import CertificationEvidence
8
+
9
+
10
+ CODEX_01540_RESPONSES_KEY = CapabilityKey(
11
+ "codex",
12
+ "responses",
13
+ "openai-compatible",
14
+ client_version="0.154.0",
15
+ )
16
+
17
+
18
+ @dataclass(frozen=True, slots=True)
19
+ class CodexRecertArea:
20
+ area: str
21
+ state: CompatibilityState
22
+ evidence_id: str
23
+ reason: str
24
+
25
+ def to_primitive(self) -> dict[str, str]:
26
+ return {
27
+ "area": self.area,
28
+ "state": self.state.value,
29
+ "evidence_id": self.evidence_id,
30
+ "reason": self.reason,
31
+ }
32
+
33
+
34
+ @dataclass(frozen=True, slots=True)
35
+ class Codex01540RecertificationBundle:
36
+ profile: CapabilityProfile
37
+ record: CompatibilityRecord
38
+ evidence: CertificationEvidence
39
+ areas: tuple[CodexRecertArea, ...]
40
+ detector: CapabilityDetector
41
+
42
+ def to_primitive(self) -> dict[str, object]:
43
+ return {
44
+ "profile": self.profile.to_primitive(),
45
+ "record": {
46
+ "state": self.record.state.value,
47
+ "evidence_id": self.record.evidence_id,
48
+ "reason": self.record.reason,
49
+ },
50
+ "evidence": self.evidence.to_primitive(),
51
+ "areas": [item.to_primitive() for item in self.areas],
52
+ }
53
+
54
+
55
+ def _area(area: str, state: CompatibilityState, reason: str) -> CodexRecertArea:
56
+ return CodexRecertArea(
57
+ area=area,
58
+ state=state,
59
+ evidence_id="token-codex-01540-recert-1",
60
+ reason=reason,
61
+ )
62
+
63
+
64
+ def build_codex_01540_recertification() -> Codex01540RecertificationBundle:
65
+ evidence = CertificationEvidence(
66
+ evidence_id="token-codex-01540-recert-1:responses-boundary",
67
+ version_scope="codex-cli-0.154.0",
68
+ scope="codex-responses/request-response-boundary",
69
+ assertion_ids=(
70
+ "exact-binary-checksum",
71
+ "responses-direct-loopback",
72
+ "responses-via-token-loopback",
73
+ "responses-request-byte-replay",
74
+ "cached-token-response-byte-fidelity",
75
+ "resume-history-continuity",
76
+ "fork-history-continuity",
77
+ ),
78
+ )
79
+ profile = CapabilityProfile(
80
+ key=CODEX_01540_RESPONSES_KEY,
81
+ tool_calls=True,
82
+ reasoning_state="opaque-preserved",
83
+ streaming=True,
84
+ exact_byte_preservation=True,
85
+ evidence_id=evidence.evidence_id,
86
+ )
87
+ record = CompatibilityRecord(
88
+ key=CODEX_01540_RESPONSES_KEY,
89
+ state=CompatibilityState.CERTIFIED,
90
+ evidence_id=evidence.evidence_id,
91
+ reason="request_boundary_certified",
92
+ )
93
+ capabilities = CapabilityRegistry()
94
+ compatibility = CompatibilityRegistry()
95
+ capabilities.register(profile)
96
+ compatibility.register(record)
97
+ areas = (
98
+ _area("responses_request_fidelity", CompatibilityState.CERTIFIED, "exact_loopback_replay"),
99
+ _area("responses_response_fidelity", CompatibilityState.CERTIFIED, "opaque_stream_forwarding"),
100
+ _area("token_off_on_boundary", CompatibilityState.CERTIFIED, "exact_request_boundary_observed"),
101
+ _area("native_context_compaction", CompatibilityState.PASSTHROUGH_ONLY, "native_internal_state_not_observed_by_token"),
102
+ _area("authorization_across_compaction", CompatibilityState.PASSTHROUGH_ONLY, "native_authorization_state_not_observed_by_token"),
103
+ _area("resume_fork_continuity", CompatibilityState.CERTIFIED, "exact_cli_history_continuity_observed"),
104
+ _area("long_context_continuity", CompatibilityState.PASSTHROUGH_ONLY, "long_context_compaction_semantics_not_observed_end_to_end"),
105
+ _area("dynamic_plugin_skill_refresh", CompatibilityState.PASSTHROUGH_ONLY, "dynamic_session_capabilities_not_consumed_by_token"),
106
+ _area("mcp_oauth_rejected_no_replay", CompatibilityState.PASSTHROUGH_ONLY, "mcp_transport_outside_token_gateway_boundary"),
107
+ _area("prompt_server_cache_semantics", CompatibilityState.PASSTHROUGH_ONLY, "provider_cache_and_billing_not_observed_locally"),
108
+ )
109
+ return Codex01540RecertificationBundle(
110
+ profile=profile,
111
+ record=record,
112
+ evidence=evidence,
113
+ areas=areas,
114
+ detector=CapabilityDetector(capabilities, compatibility),
115
+ )