tokenbiryani 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. tokenbiryani/__init__.py +3 -0
  2. tokenbiryani/api/__init__.py +0 -0
  3. tokenbiryani/api/app.py +583 -0
  4. tokenbiryani/api/asgi.py +32 -0
  5. tokenbiryani/cli.py +1045 -0
  6. tokenbiryani/config.py +532 -0
  7. tokenbiryani/core/__init__.py +0 -0
  8. tokenbiryani/core/account.py +258 -0
  9. tokenbiryani/core/batch.py +135 -0
  10. tokenbiryani/core/breaker.py +53 -0
  11. tokenbiryani/core/cacheadvice.py +239 -0
  12. tokenbiryani/core/diagnostics.py +131 -0
  13. tokenbiryani/core/estimator.py +180 -0
  14. tokenbiryani/core/gateway.py +2395 -0
  15. tokenbiryani/core/handoff.py +87 -0
  16. tokenbiryani/core/keys.py +199 -0
  17. tokenbiryani/core/limits.py +440 -0
  18. tokenbiryani/core/oauth.py +222 -0
  19. tokenbiryani/core/pacing.py +320 -0
  20. tokenbiryani/core/queue.py +132 -0
  21. tokenbiryani/core/router.py +323 -0
  22. tokenbiryani/core/secrets.py +114 -0
  23. tokenbiryani/core/session.py +117 -0
  24. tokenbiryani/dashboard/__init__.py +56 -0
  25. tokenbiryani/dashboard/console.css +610 -0
  26. tokenbiryani/dashboard/console.html +3250 -0
  27. tokenbiryani/observability/__init__.py +0 -0
  28. tokenbiryani/observability/events.py +171 -0
  29. tokenbiryani/observability/usage.py +226 -0
  30. tokenbiryani/prices.yaml +77 -0
  31. tokenbiryani/providers/__init__.py +0 -0
  32. tokenbiryani/providers/anthropic_api.py +118 -0
  33. tokenbiryani/providers/base.py +173 -0
  34. tokenbiryani/providers/bedrock.py +182 -0
  35. tokenbiryani/providers/oauth.py +165 -0
  36. tokenbiryani/providers/oauth_credentials.py +293 -0
  37. tokenbiryani/providers/translate.py +35 -0
  38. tokenbiryani/providers/vertex.py +144 -0
  39. tokenbiryani/proxy/__init__.py +0 -0
  40. tokenbiryani/proxy/errors.py +169 -0
  41. tokenbiryani/proxy/sse.py +98 -0
  42. tokenbiryani/store/__init__.py +0 -0
  43. tokenbiryani/store/base.py +150 -0
  44. tokenbiryani/store/memory.py +149 -0
  45. tokenbiryani/store/redis_store.py +222 -0
  46. tokenbiryani/store/sqlite.py +336 -0
  47. tokenbiryani-0.2.0.dist-info/METADATA +697 -0
  48. tokenbiryani-0.2.0.dist-info/RECORD +51 -0
  49. tokenbiryani-0.2.0.dist-info/WHEEL +4 -0
  50. tokenbiryani-0.2.0.dist-info/entry_points.txt +2 -0
  51. tokenbiryani-0.2.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,131 @@
1
+ """What the rate-limit headers on a real response actually say.
2
+
3
+ This is the one check no mock-backed test can make: every test in this project runs
4
+ against the suite's fake upstream, which encodes an assumption about how
5
+ the upstream spells its headers. If the real API spells one differently, that window
6
+ stays empty, the account reads as full, and the router silently degrades to
7
+ round-robin — shredding the prompt cache while every screen looks healthy.
8
+
9
+ It lives here, rather than inside `cmd_doctor`, because two callers need the same
10
+ answer from the same list: `tokenbiryani doctor` in the terminal, and the console's
11
+ verify step. Two copies of the expected-header list is exactly the drift the check
12
+ exists to catch.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import time
18
+ from typing import Any, Dict, List, Mapping
19
+
20
+ from .limits import UNIFIED_WINDOWS, LimitMirror
21
+
22
+ #: The three windows the limit mirror models, in the spelling the upstream uses.
23
+ LIMIT_WINDOWS = ["requests", "input-tokens", "output-tokens"]
24
+
25
+ #: ...and the three facts it needs about each of them.
26
+ HEADER_TEMPLATES = [
27
+ "anthropic-ratelimit-{}-limit",
28
+ "anthropic-ratelimit-{}-remaining",
29
+ "anthropic-ratelimit-{}-reset",
30
+ ]
31
+
32
+ #: A subscription session answers a different question. It never sends the triples
33
+ #: above; it sends how much of each rolling window is spent. An account reporting
34
+ #: these is not broken, so checking it against the API-key list would report nine
35
+ #: faults where there are none — which is the exact false alarm this module exists
36
+ #: to prevent, pointed the other way.
37
+ UNIFIED_TEMPLATES = [
38
+ "anthropic-ratelimit-unified-{}-status",
39
+ "anthropic-ratelimit-unified-{}-utilization",
40
+ "anthropic-ratelimit-unified-{}-reset",
41
+ ]
42
+
43
+
44
+ def expected_headers(unified: bool = False) -> List[str]:
45
+ """Every header the router depends on, in the order a report should show them."""
46
+ if unified:
47
+ return [
48
+ template.format(window)
49
+ for window in UNIFIED_WINDOWS
50
+ for template in UNIFIED_TEMPLATES
51
+ ]
52
+ return [
53
+ template.format(window)
54
+ for window in LIMIT_WINDOWS
55
+ for template in HEADER_TEMPLATES
56
+ ]
57
+
58
+
59
+ def looks_unified(lowered: Mapping[str, str]) -> bool:
60
+ """True when the response is a subscription session's, not an API key's."""
61
+ return any(
62
+ str(name).startswith("anthropic-ratelimit-unified-") for name in lowered
63
+ )
64
+
65
+
66
+ def inspect_headers(headers: Mapping[str, str], now: float = 0.0) -> Dict[str, Any]:
67
+ """Compare what arrived against what the mirror looks for, and parse it.
68
+
69
+ Returns the whole picture rather than a verdict, because the useful thing to show
70
+ a human is the mismatch itself: which header was expected, what came instead, and
71
+ what the router would therefore believe.
72
+ """
73
+ now = now or time.time()
74
+ lowered = {str(k).lower(): str(v) for k, v in dict(headers).items()}
75
+ unified = looks_unified(lowered)
76
+
77
+ checked = []
78
+ missing = []
79
+ for name in expected_headers(unified):
80
+ present = name in lowered
81
+ if not present:
82
+ missing.append(name)
83
+ checked.append({"header": name, "present": present, "value": lowered.get(name)})
84
+
85
+ # The real proof is not that the keys exist but that the mirror the router reads
86
+ # parsed something out of them.
87
+ mirror = LimitMirror(observable=not unified)
88
+ mirror.update_from_headers(dict(headers), now)
89
+ snapshot = mirror.snapshot(now)
90
+ if unified:
91
+ parsed = {
92
+ f"unified_{name}": snapshot["unified"][name]
93
+ for name in snapshot["unified"]
94
+ }
95
+ else:
96
+ parsed = {
97
+ label: {
98
+ "limit": snapshot[label]["limit"],
99
+ "remaining": snapshot[label]["remaining"],
100
+ "reset_in": snapshot[label]["reset_in"],
101
+ }
102
+ for label in ("requests", "input_tokens", "output_tokens")
103
+ }
104
+
105
+ if not missing:
106
+ consequence = ""
107
+ elif unified:
108
+ consequence = (
109
+ f"{len(missing)} of {len(expected_headers(True))} unified headers are "
110
+ "missing or spelled differently. This account routes on assumed headroom "
111
+ "instead of its real utilisation, so a nearly-exhausted subscription "
112
+ "still reads as half full."
113
+ )
114
+ else:
115
+ consequence = (
116
+ f"{len(missing)} of {len(expected_headers())} headers are missing or "
117
+ "spelled differently. Those windows stay empty, so this account reads "
118
+ "as full and routing degrades to round-robin — which shreds the prompt "
119
+ "cache while every meter looks healthy."
120
+ )
121
+
122
+ return {
123
+ "ok": not missing,
124
+ "family": "unified" if unified else "limits",
125
+ "headroom": snapshot["headroom"],
126
+ "checked": checked,
127
+ "missing": missing,
128
+ "returned": {k: v for k, v in sorted(lowered.items()) if k.startswith("anthropic-")},
129
+ "parsed": parsed,
130
+ "consequence": consequence,
131
+ }
@@ -0,0 +1,180 @@
1
+ """Predicting what a request will actually return, instead of believing its ceiling.
2
+
3
+ `max_tokens` is the only honest upper bound available before a request runs, which
4
+ is why the lease started out using it. It is also wildly wrong in the case this
5
+ gateway exists to serve: an agent client sends `max_tokens: 32000` and returns a few
6
+ hundred tokens.
7
+
8
+ The lease is released afterwards, so no quota is *spent* on the difference. The cost
9
+ is paid during the request instead. An account whose output window is leased at 32k
10
+ reads as full, an account that reads as full is filtered out of routing, and a
11
+ conversation whose owner is filtered out gets re-homed onto a credential that has
12
+ never seen its prefix. That is how an estimation problem turns into a cache break,
13
+ which is the most expensive thing that can happen here.
14
+
15
+ Measured against a mirrored output window, believing the ceiling costs most of the
16
+ pool's concurrency:
17
+
18
+ output limit 16,000/window, max_tokens 8,192 -> 1 concurrent request
19
+ output limit 64,000/window, max_tokens 32,000 -> 2 concurrent requests
20
+
21
+ So: keep a rolling sample of what each model really returns, lease a high quantile
22
+ of it, and never lease more than the caller's own ceiling — the prediction can only
23
+ ever reserve *less* than today. Fall back to the ceiling until there is enough
24
+ evidence to beat it, and again if the prediction turns out to be beaten too often.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ from collections import defaultdict, deque
30
+ from typing import Any, Deque, Dict
31
+
32
+ #: Estimating from fewer samples than this is guessing with extra steps.
33
+ DEFAULT_MIN_SAMPLES = 20
34
+
35
+ #: How many recent completions per model take part. Long enough to be stable across
36
+ #: a mix of short and long turns, short enough to follow a change in how a client
37
+ #: is being used.
38
+ DEFAULT_WINDOW = 200
39
+
40
+ #: Never predict below this. A model that has only ever answered in one word must
41
+ #: not leave a two-word answer under-reserved.
42
+ DEFAULT_FLOOR = 256
43
+
44
+ #: A p95 predictor is beaten about 5% of the time by construction. This is the
45
+ #: safety valve for a distribution that has changed shape, not for that.
46
+ DEFAULT_MAX_UNDERSHOOT = 0.20
47
+
48
+ MODE_ADAPTIVE = "adaptive"
49
+ MODE_CEILING = "max_tokens"
50
+
51
+
52
+ class OutputEstimator:
53
+ """Per-model output-size prediction, bounded by the caller's own `max_tokens`.
54
+
55
+ Shared across the pool rather than held per account: what a model returns is a
56
+ property of the model and the traffic, not of the credential it was billed to,
57
+ and splitting the samples per account would just make each one slower to learn
58
+ the same number.
59
+ """
60
+
61
+ def __init__(
62
+ self,
63
+ mode: str = MODE_ADAPTIVE,
64
+ quantile: float = 0.95,
65
+ min_samples: int = DEFAULT_MIN_SAMPLES,
66
+ window: int = DEFAULT_WINDOW,
67
+ floor: int = DEFAULT_FLOOR,
68
+ max_undershoot: float = DEFAULT_MAX_UNDERSHOOT,
69
+ ) -> None:
70
+ self.mode = mode
71
+ self.quantile = min(1.0, max(0.5, float(quantile)))
72
+ self.min_samples = max(1, int(min_samples))
73
+ self.window = max(self.min_samples, int(window))
74
+ self.floor = max(1, int(floor))
75
+ self.max_undershoot = max(0.0, min(1.0, float(max_undershoot)))
76
+ self._samples: Dict[str, Deque[int]] = defaultdict(
77
+ lambda: deque(maxlen=self.window)
78
+ )
79
+ self._undershoots: Dict[str, Deque[bool]] = defaultdict(
80
+ lambda: deque(maxlen=self.window)
81
+ )
82
+
83
+ @property
84
+ def enabled(self) -> bool:
85
+ return self.mode == MODE_ADAPTIVE
86
+
87
+ def reconfigure(
88
+ self,
89
+ mode: str,
90
+ quantile: float,
91
+ min_samples: int,
92
+ window: int,
93
+ floor: int,
94
+ max_undershoot: float,
95
+ ) -> None:
96
+ """Apply new settings without forgetting what has been learned.
97
+
98
+ A config reload must not send the pool back to leasing ceilings for the next
99
+ twenty requests of every model. The samples are observed state, not
100
+ configuration, and they survive exactly as managed keys do.
101
+ """
102
+ self.mode = mode
103
+ self.quantile = min(1.0, max(0.5, float(quantile)))
104
+ self.min_samples = max(1, int(min_samples))
105
+ self.floor = max(1, int(floor))
106
+ self.max_undershoot = max(0.0, min(1.0, float(max_undershoot)))
107
+ new_window = max(self.min_samples, int(window))
108
+ if new_window != self.window:
109
+ self.window = new_window
110
+ for model, sizes in list(self._samples.items()):
111
+ self._samples[model] = deque(sizes, maxlen=new_window)
112
+ for model, missed in list(self._undershoots.items()):
113
+ self._undershoots[model] = deque(missed, maxlen=new_window)
114
+
115
+ def predict(self, model: str, ceiling: int) -> int:
116
+ """What to lease for this request. Never more than `ceiling`.
117
+
118
+ Returning `ceiling` is always a valid answer and is what happens whenever
119
+ the estimator has no business having an opinion.
120
+ """
121
+ ceiling = max(1, int(ceiling))
122
+ if not self.enabled:
123
+ return ceiling
124
+ samples = self._samples.get(model)
125
+ if samples is None or len(samples) < self.min_samples:
126
+ return ceiling
127
+ if self._undershoot_rate(model) > self.max_undershoot:
128
+ # The distribution moved. Stop predicting until the record of being
129
+ # beaten ages out of the window.
130
+ return ceiling
131
+ predicted = _quantile(samples, self.quantile)
132
+ return max(self.floor, min(ceiling, predicted))
133
+
134
+ def observe(self, model: str, predicted: int, actual: int) -> None:
135
+ """Record one completion: what it returned, and whether it beat the lease."""
136
+ if actual <= 0:
137
+ # No usage came back — an error, or an upstream that reported none.
138
+ # Neither says anything about how long an answer is.
139
+ return
140
+ self._samples[model].append(int(actual))
141
+ self._undershoots[model].append(int(actual) > int(predicted))
142
+
143
+ def _undershoot_rate(self, model: str) -> float:
144
+ seen = self._undershoots.get(model)
145
+ if not seen or len(seen) < self.min_samples:
146
+ return 0.0
147
+ return sum(1 for missed in seen if missed) / float(len(seen))
148
+
149
+ def snapshot(self) -> Dict[str, Any]:
150
+ """What the estimator currently believes, for the admin API and the console."""
151
+ models: Dict[str, Any] = {}
152
+ for model, samples in self._samples.items():
153
+ if not samples:
154
+ continue
155
+ ordered = sorted(samples)
156
+ models[model] = {
157
+ "samples": len(ordered),
158
+ "predicting": len(ordered) >= self.min_samples
159
+ and self._undershoot_rate(model) <= self.max_undershoot,
160
+ "median": ordered[len(ordered) // 2],
161
+ "p95": _quantile(samples, 0.95),
162
+ "max": ordered[-1],
163
+ "undershoot_rate": round(self._undershoot_rate(model), 4),
164
+ }
165
+ return {
166
+ "mode": self.mode,
167
+ "quantile": self.quantile,
168
+ "min_samples": self.min_samples,
169
+ "floor": self.floor,
170
+ "models": models,
171
+ }
172
+
173
+
174
+ def _quantile(values: Any, fraction: float) -> int:
175
+ """Nearest-rank quantile. No numpy, and no interpolation to argue about."""
176
+ ordered = sorted(values)
177
+ if not ordered:
178
+ return 0
179
+ index = min(len(ordered) - 1, int(round(fraction * (len(ordered) - 1))))
180
+ return int(ordered[index])