tokenbiryani 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokenbiryani/__init__.py +3 -0
- tokenbiryani/api/__init__.py +0 -0
- tokenbiryani/api/app.py +583 -0
- tokenbiryani/api/asgi.py +32 -0
- tokenbiryani/cli.py +1045 -0
- tokenbiryani/config.py +532 -0
- tokenbiryani/core/__init__.py +0 -0
- tokenbiryani/core/account.py +258 -0
- tokenbiryani/core/batch.py +135 -0
- tokenbiryani/core/breaker.py +53 -0
- tokenbiryani/core/cacheadvice.py +239 -0
- tokenbiryani/core/diagnostics.py +131 -0
- tokenbiryani/core/estimator.py +180 -0
- tokenbiryani/core/gateway.py +2395 -0
- tokenbiryani/core/handoff.py +87 -0
- tokenbiryani/core/keys.py +199 -0
- tokenbiryani/core/limits.py +440 -0
- tokenbiryani/core/oauth.py +222 -0
- tokenbiryani/core/pacing.py +320 -0
- tokenbiryani/core/queue.py +132 -0
- tokenbiryani/core/router.py +323 -0
- tokenbiryani/core/secrets.py +114 -0
- tokenbiryani/core/session.py +117 -0
- tokenbiryani/dashboard/__init__.py +56 -0
- tokenbiryani/dashboard/console.css +610 -0
- tokenbiryani/dashboard/console.html +3250 -0
- tokenbiryani/observability/__init__.py +0 -0
- tokenbiryani/observability/events.py +171 -0
- tokenbiryani/observability/usage.py +226 -0
- tokenbiryani/prices.yaml +77 -0
- tokenbiryani/providers/__init__.py +0 -0
- tokenbiryani/providers/anthropic_api.py +118 -0
- tokenbiryani/providers/base.py +173 -0
- tokenbiryani/providers/bedrock.py +182 -0
- tokenbiryani/providers/oauth.py +165 -0
- tokenbiryani/providers/oauth_credentials.py +293 -0
- tokenbiryani/providers/translate.py +35 -0
- tokenbiryani/providers/vertex.py +144 -0
- tokenbiryani/proxy/__init__.py +0 -0
- tokenbiryani/proxy/errors.py +169 -0
- tokenbiryani/proxy/sse.py +98 -0
- tokenbiryani/store/__init__.py +0 -0
- tokenbiryani/store/base.py +150 -0
- tokenbiryani/store/memory.py +149 -0
- tokenbiryani/store/redis_store.py +222 -0
- tokenbiryani/store/sqlite.py +336 -0
- tokenbiryani-0.2.0.dist-info/METADATA +697 -0
- tokenbiryani-0.2.0.dist-info/RECORD +51 -0
- tokenbiryani-0.2.0.dist-info/WHEEL +4 -0
- tokenbiryani-0.2.0.dist-info/entry_points.txt +2 -0
- tokenbiryani-0.2.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""What the rate-limit headers on a real response actually say.
|
|
2
|
+
|
|
3
|
+
This is the one check no mock-backed test can make: every test in this project runs
|
|
4
|
+
against the suite's fake upstream, which encodes an assumption about how
|
|
5
|
+
the upstream spells its headers. If the real API spells one differently, that window
|
|
6
|
+
stays empty, the account reads as full, and the router silently degrades to
|
|
7
|
+
round-robin — shredding the prompt cache while every screen looks healthy.
|
|
8
|
+
|
|
9
|
+
It lives here, rather than inside `cmd_doctor`, because two callers need the same
|
|
10
|
+
answer from the same list: `tokenbiryani doctor` in the terminal, and the console's
|
|
11
|
+
verify step. Two copies of the expected-header list is exactly the drift the check
|
|
12
|
+
exists to catch.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import time
|
|
18
|
+
from typing import Any, Dict, List, Mapping
|
|
19
|
+
|
|
20
|
+
from .limits import UNIFIED_WINDOWS, LimitMirror
|
|
21
|
+
|
|
22
|
+
#: The three windows the limit mirror models, in the spelling the upstream uses.
|
|
23
|
+
LIMIT_WINDOWS = ["requests", "input-tokens", "output-tokens"]
|
|
24
|
+
|
|
25
|
+
#: ...and the three facts it needs about each of them.
|
|
26
|
+
HEADER_TEMPLATES = [
|
|
27
|
+
"anthropic-ratelimit-{}-limit",
|
|
28
|
+
"anthropic-ratelimit-{}-remaining",
|
|
29
|
+
"anthropic-ratelimit-{}-reset",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
#: A subscription session answers a different question. It never sends the triples
|
|
33
|
+
#: above; it sends how much of each rolling window is spent. An account reporting
|
|
34
|
+
#: these is not broken, so checking it against the API-key list would report nine
|
|
35
|
+
#: faults where there are none — which is the exact false alarm this module exists
|
|
36
|
+
#: to prevent, pointed the other way.
|
|
37
|
+
UNIFIED_TEMPLATES = [
|
|
38
|
+
"anthropic-ratelimit-unified-{}-status",
|
|
39
|
+
"anthropic-ratelimit-unified-{}-utilization",
|
|
40
|
+
"anthropic-ratelimit-unified-{}-reset",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def expected_headers(unified: bool = False) -> List[str]:
|
|
45
|
+
"""Every header the router depends on, in the order a report should show them."""
|
|
46
|
+
if unified:
|
|
47
|
+
return [
|
|
48
|
+
template.format(window)
|
|
49
|
+
for window in UNIFIED_WINDOWS
|
|
50
|
+
for template in UNIFIED_TEMPLATES
|
|
51
|
+
]
|
|
52
|
+
return [
|
|
53
|
+
template.format(window)
|
|
54
|
+
for window in LIMIT_WINDOWS
|
|
55
|
+
for template in HEADER_TEMPLATES
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def looks_unified(lowered: Mapping[str, str]) -> bool:
|
|
60
|
+
"""True when the response is a subscription session's, not an API key's."""
|
|
61
|
+
return any(
|
|
62
|
+
str(name).startswith("anthropic-ratelimit-unified-") for name in lowered
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def inspect_headers(headers: Mapping[str, str], now: float = 0.0) -> Dict[str, Any]:
|
|
67
|
+
"""Compare what arrived against what the mirror looks for, and parse it.
|
|
68
|
+
|
|
69
|
+
Returns the whole picture rather than a verdict, because the useful thing to show
|
|
70
|
+
a human is the mismatch itself: which header was expected, what came instead, and
|
|
71
|
+
what the router would therefore believe.
|
|
72
|
+
"""
|
|
73
|
+
now = now or time.time()
|
|
74
|
+
lowered = {str(k).lower(): str(v) for k, v in dict(headers).items()}
|
|
75
|
+
unified = looks_unified(lowered)
|
|
76
|
+
|
|
77
|
+
checked = []
|
|
78
|
+
missing = []
|
|
79
|
+
for name in expected_headers(unified):
|
|
80
|
+
present = name in lowered
|
|
81
|
+
if not present:
|
|
82
|
+
missing.append(name)
|
|
83
|
+
checked.append({"header": name, "present": present, "value": lowered.get(name)})
|
|
84
|
+
|
|
85
|
+
# The real proof is not that the keys exist but that the mirror the router reads
|
|
86
|
+
# parsed something out of them.
|
|
87
|
+
mirror = LimitMirror(observable=not unified)
|
|
88
|
+
mirror.update_from_headers(dict(headers), now)
|
|
89
|
+
snapshot = mirror.snapshot(now)
|
|
90
|
+
if unified:
|
|
91
|
+
parsed = {
|
|
92
|
+
f"unified_{name}": snapshot["unified"][name]
|
|
93
|
+
for name in snapshot["unified"]
|
|
94
|
+
}
|
|
95
|
+
else:
|
|
96
|
+
parsed = {
|
|
97
|
+
label: {
|
|
98
|
+
"limit": snapshot[label]["limit"],
|
|
99
|
+
"remaining": snapshot[label]["remaining"],
|
|
100
|
+
"reset_in": snapshot[label]["reset_in"],
|
|
101
|
+
}
|
|
102
|
+
for label in ("requests", "input_tokens", "output_tokens")
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
if not missing:
|
|
106
|
+
consequence = ""
|
|
107
|
+
elif unified:
|
|
108
|
+
consequence = (
|
|
109
|
+
f"{len(missing)} of {len(expected_headers(True))} unified headers are "
|
|
110
|
+
"missing or spelled differently. This account routes on assumed headroom "
|
|
111
|
+
"instead of its real utilisation, so a nearly-exhausted subscription "
|
|
112
|
+
"still reads as half full."
|
|
113
|
+
)
|
|
114
|
+
else:
|
|
115
|
+
consequence = (
|
|
116
|
+
f"{len(missing)} of {len(expected_headers())} headers are missing or "
|
|
117
|
+
"spelled differently. Those windows stay empty, so this account reads "
|
|
118
|
+
"as full and routing degrades to round-robin — which shreds the prompt "
|
|
119
|
+
"cache while every meter looks healthy."
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
return {
|
|
123
|
+
"ok": not missing,
|
|
124
|
+
"family": "unified" if unified else "limits",
|
|
125
|
+
"headroom": snapshot["headroom"],
|
|
126
|
+
"checked": checked,
|
|
127
|
+
"missing": missing,
|
|
128
|
+
"returned": {k: v for k, v in sorted(lowered.items()) if k.startswith("anthropic-")},
|
|
129
|
+
"parsed": parsed,
|
|
130
|
+
"consequence": consequence,
|
|
131
|
+
}
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Predicting what a request will actually return, instead of believing its ceiling.
|
|
2
|
+
|
|
3
|
+
`max_tokens` is the only honest upper bound available before a request runs, which
|
|
4
|
+
is why the lease started out using it. It is also wildly wrong in the case this
|
|
5
|
+
gateway exists to serve: an agent client sends `max_tokens: 32000` and returns a few
|
|
6
|
+
hundred tokens.
|
|
7
|
+
|
|
8
|
+
The lease is released afterwards, so no quota is *spent* on the difference. The cost
|
|
9
|
+
is paid during the request instead. An account whose output window is leased at 32k
|
|
10
|
+
reads as full, an account that reads as full is filtered out of routing, and a
|
|
11
|
+
conversation whose owner is filtered out gets re-homed onto a credential that has
|
|
12
|
+
never seen its prefix. That is how an estimation problem turns into a cache break,
|
|
13
|
+
which is the most expensive thing that can happen here.
|
|
14
|
+
|
|
15
|
+
Measured against a mirrored output window, believing the ceiling costs most of the
|
|
16
|
+
pool's concurrency:
|
|
17
|
+
|
|
18
|
+
output limit 16,000/window, max_tokens 8,192 -> 1 concurrent request
|
|
19
|
+
output limit 64,000/window, max_tokens 32,000 -> 2 concurrent requests
|
|
20
|
+
|
|
21
|
+
So: keep a rolling sample of what each model really returns, lease a high quantile
|
|
22
|
+
of it, and never lease more than the caller's own ceiling — the prediction can only
|
|
23
|
+
ever reserve *less* than today. Fall back to the ceiling until there is enough
|
|
24
|
+
evidence to beat it, and again if the prediction turns out to be beaten too often.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
from collections import defaultdict, deque
|
|
30
|
+
from typing import Any, Deque, Dict
|
|
31
|
+
|
|
32
|
+
#: Estimating from fewer samples than this is guessing with extra steps.
|
|
33
|
+
DEFAULT_MIN_SAMPLES = 20
|
|
34
|
+
|
|
35
|
+
#: How many recent completions per model take part. Long enough to be stable across
|
|
36
|
+
#: a mix of short and long turns, short enough to follow a change in how a client
|
|
37
|
+
#: is being used.
|
|
38
|
+
DEFAULT_WINDOW = 200
|
|
39
|
+
|
|
40
|
+
#: Never predict below this. A model that has only ever answered in one word must
|
|
41
|
+
#: not leave a two-word answer under-reserved.
|
|
42
|
+
DEFAULT_FLOOR = 256
|
|
43
|
+
|
|
44
|
+
#: A p95 predictor is beaten about 5% of the time by construction. This is the
|
|
45
|
+
#: safety valve for a distribution that has changed shape, not for that.
|
|
46
|
+
DEFAULT_MAX_UNDERSHOOT = 0.20
|
|
47
|
+
|
|
48
|
+
MODE_ADAPTIVE = "adaptive"
|
|
49
|
+
MODE_CEILING = "max_tokens"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class OutputEstimator:
|
|
53
|
+
"""Per-model output-size prediction, bounded by the caller's own `max_tokens`.
|
|
54
|
+
|
|
55
|
+
Shared across the pool rather than held per account: what a model returns is a
|
|
56
|
+
property of the model and the traffic, not of the credential it was billed to,
|
|
57
|
+
and splitting the samples per account would just make each one slower to learn
|
|
58
|
+
the same number.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
mode: str = MODE_ADAPTIVE,
|
|
64
|
+
quantile: float = 0.95,
|
|
65
|
+
min_samples: int = DEFAULT_MIN_SAMPLES,
|
|
66
|
+
window: int = DEFAULT_WINDOW,
|
|
67
|
+
floor: int = DEFAULT_FLOOR,
|
|
68
|
+
max_undershoot: float = DEFAULT_MAX_UNDERSHOOT,
|
|
69
|
+
) -> None:
|
|
70
|
+
self.mode = mode
|
|
71
|
+
self.quantile = min(1.0, max(0.5, float(quantile)))
|
|
72
|
+
self.min_samples = max(1, int(min_samples))
|
|
73
|
+
self.window = max(self.min_samples, int(window))
|
|
74
|
+
self.floor = max(1, int(floor))
|
|
75
|
+
self.max_undershoot = max(0.0, min(1.0, float(max_undershoot)))
|
|
76
|
+
self._samples: Dict[str, Deque[int]] = defaultdict(
|
|
77
|
+
lambda: deque(maxlen=self.window)
|
|
78
|
+
)
|
|
79
|
+
self._undershoots: Dict[str, Deque[bool]] = defaultdict(
|
|
80
|
+
lambda: deque(maxlen=self.window)
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
@property
|
|
84
|
+
def enabled(self) -> bool:
|
|
85
|
+
return self.mode == MODE_ADAPTIVE
|
|
86
|
+
|
|
87
|
+
def reconfigure(
|
|
88
|
+
self,
|
|
89
|
+
mode: str,
|
|
90
|
+
quantile: float,
|
|
91
|
+
min_samples: int,
|
|
92
|
+
window: int,
|
|
93
|
+
floor: int,
|
|
94
|
+
max_undershoot: float,
|
|
95
|
+
) -> None:
|
|
96
|
+
"""Apply new settings without forgetting what has been learned.
|
|
97
|
+
|
|
98
|
+
A config reload must not send the pool back to leasing ceilings for the next
|
|
99
|
+
twenty requests of every model. The samples are observed state, not
|
|
100
|
+
configuration, and they survive exactly as managed keys do.
|
|
101
|
+
"""
|
|
102
|
+
self.mode = mode
|
|
103
|
+
self.quantile = min(1.0, max(0.5, float(quantile)))
|
|
104
|
+
self.min_samples = max(1, int(min_samples))
|
|
105
|
+
self.floor = max(1, int(floor))
|
|
106
|
+
self.max_undershoot = max(0.0, min(1.0, float(max_undershoot)))
|
|
107
|
+
new_window = max(self.min_samples, int(window))
|
|
108
|
+
if new_window != self.window:
|
|
109
|
+
self.window = new_window
|
|
110
|
+
for model, sizes in list(self._samples.items()):
|
|
111
|
+
self._samples[model] = deque(sizes, maxlen=new_window)
|
|
112
|
+
for model, missed in list(self._undershoots.items()):
|
|
113
|
+
self._undershoots[model] = deque(missed, maxlen=new_window)
|
|
114
|
+
|
|
115
|
+
def predict(self, model: str, ceiling: int) -> int:
|
|
116
|
+
"""What to lease for this request. Never more than `ceiling`.
|
|
117
|
+
|
|
118
|
+
Returning `ceiling` is always a valid answer and is what happens whenever
|
|
119
|
+
the estimator has no business having an opinion.
|
|
120
|
+
"""
|
|
121
|
+
ceiling = max(1, int(ceiling))
|
|
122
|
+
if not self.enabled:
|
|
123
|
+
return ceiling
|
|
124
|
+
samples = self._samples.get(model)
|
|
125
|
+
if samples is None or len(samples) < self.min_samples:
|
|
126
|
+
return ceiling
|
|
127
|
+
if self._undershoot_rate(model) > self.max_undershoot:
|
|
128
|
+
# The distribution moved. Stop predicting until the record of being
|
|
129
|
+
# beaten ages out of the window.
|
|
130
|
+
return ceiling
|
|
131
|
+
predicted = _quantile(samples, self.quantile)
|
|
132
|
+
return max(self.floor, min(ceiling, predicted))
|
|
133
|
+
|
|
134
|
+
def observe(self, model: str, predicted: int, actual: int) -> None:
|
|
135
|
+
"""Record one completion: what it returned, and whether it beat the lease."""
|
|
136
|
+
if actual <= 0:
|
|
137
|
+
# No usage came back — an error, or an upstream that reported none.
|
|
138
|
+
# Neither says anything about how long an answer is.
|
|
139
|
+
return
|
|
140
|
+
self._samples[model].append(int(actual))
|
|
141
|
+
self._undershoots[model].append(int(actual) > int(predicted))
|
|
142
|
+
|
|
143
|
+
def _undershoot_rate(self, model: str) -> float:
|
|
144
|
+
seen = self._undershoots.get(model)
|
|
145
|
+
if not seen or len(seen) < self.min_samples:
|
|
146
|
+
return 0.0
|
|
147
|
+
return sum(1 for missed in seen if missed) / float(len(seen))
|
|
148
|
+
|
|
149
|
+
def snapshot(self) -> Dict[str, Any]:
|
|
150
|
+
"""What the estimator currently believes, for the admin API and the console."""
|
|
151
|
+
models: Dict[str, Any] = {}
|
|
152
|
+
for model, samples in self._samples.items():
|
|
153
|
+
if not samples:
|
|
154
|
+
continue
|
|
155
|
+
ordered = sorted(samples)
|
|
156
|
+
models[model] = {
|
|
157
|
+
"samples": len(ordered),
|
|
158
|
+
"predicting": len(ordered) >= self.min_samples
|
|
159
|
+
and self._undershoot_rate(model) <= self.max_undershoot,
|
|
160
|
+
"median": ordered[len(ordered) // 2],
|
|
161
|
+
"p95": _quantile(samples, 0.95),
|
|
162
|
+
"max": ordered[-1],
|
|
163
|
+
"undershoot_rate": round(self._undershoot_rate(model), 4),
|
|
164
|
+
}
|
|
165
|
+
return {
|
|
166
|
+
"mode": self.mode,
|
|
167
|
+
"quantile": self.quantile,
|
|
168
|
+
"min_samples": self.min_samples,
|
|
169
|
+
"floor": self.floor,
|
|
170
|
+
"models": models,
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _quantile(values: Any, fraction: float) -> int:
|
|
175
|
+
"""Nearest-rank quantile. No numpy, and no interpolation to argue about."""
|
|
176
|
+
ordered = sorted(values)
|
|
177
|
+
if not ordered:
|
|
178
|
+
return 0
|
|
179
|
+
index = min(len(ordered) - 1, int(round(fraction * (len(ordered) - 1))))
|
|
180
|
+
return int(ordered[index])
|