tokenbiryani 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokenbiryani/__init__.py +3 -0
- tokenbiryani/api/__init__.py +0 -0
- tokenbiryani/api/app.py +583 -0
- tokenbiryani/api/asgi.py +32 -0
- tokenbiryani/cli.py +1045 -0
- tokenbiryani/config.py +532 -0
- tokenbiryani/core/__init__.py +0 -0
- tokenbiryani/core/account.py +258 -0
- tokenbiryani/core/batch.py +135 -0
- tokenbiryani/core/breaker.py +53 -0
- tokenbiryani/core/cacheadvice.py +239 -0
- tokenbiryani/core/diagnostics.py +131 -0
- tokenbiryani/core/estimator.py +180 -0
- tokenbiryani/core/gateway.py +2395 -0
- tokenbiryani/core/handoff.py +87 -0
- tokenbiryani/core/keys.py +199 -0
- tokenbiryani/core/limits.py +440 -0
- tokenbiryani/core/oauth.py +222 -0
- tokenbiryani/core/pacing.py +320 -0
- tokenbiryani/core/queue.py +132 -0
- tokenbiryani/core/router.py +323 -0
- tokenbiryani/core/secrets.py +114 -0
- tokenbiryani/core/session.py +117 -0
- tokenbiryani/dashboard/__init__.py +56 -0
- tokenbiryani/dashboard/console.css +610 -0
- tokenbiryani/dashboard/console.html +3250 -0
- tokenbiryani/observability/__init__.py +0 -0
- tokenbiryani/observability/events.py +171 -0
- tokenbiryani/observability/usage.py +226 -0
- tokenbiryani/prices.yaml +77 -0
- tokenbiryani/providers/__init__.py +0 -0
- tokenbiryani/providers/anthropic_api.py +118 -0
- tokenbiryani/providers/base.py +173 -0
- tokenbiryani/providers/bedrock.py +182 -0
- tokenbiryani/providers/oauth.py +165 -0
- tokenbiryani/providers/oauth_credentials.py +293 -0
- tokenbiryani/providers/translate.py +35 -0
- tokenbiryani/providers/vertex.py +144 -0
- tokenbiryani/proxy/__init__.py +0 -0
- tokenbiryani/proxy/errors.py +169 -0
- tokenbiryani/proxy/sse.py +98 -0
- tokenbiryani/store/__init__.py +0 -0
- tokenbiryani/store/base.py +150 -0
- tokenbiryani/store/memory.py +149 -0
- tokenbiryani/store/redis_store.py +222 -0
- tokenbiryani/store/sqlite.py +336 -0
- tokenbiryani-0.2.0.dist-info/METADATA +697 -0
- tokenbiryani-0.2.0.dist-info/RECORD +51 -0
- tokenbiryani-0.2.0.dist-info/WHEEL +4 -0
- tokenbiryani-0.2.0.dist-info/entry_points.txt +2 -0
- tokenbiryani-0.2.0.dist-info/licenses/LICENSE +202 -0
tokenbiryani/config.py
ADDED
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
"""Configuration: a single tokenbiryani.yaml, with ${ENV} interpolation and hot reload."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from typing import Any, Dict, List, Optional
|
|
9
|
+
|
|
10
|
+
import yaml
|
|
11
|
+
|
|
12
|
+
_ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)(?::-([^}]*))?\}")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ConfigError(ValueError):
|
|
16
|
+
"""Raised when tokenbiryani.yaml is missing something required or self-contradictory."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def interpolate(value: Any) -> Any:
|
|
20
|
+
"""Expand ${VAR} and ${VAR:-default} anywhere in a nested structure."""
|
|
21
|
+
if isinstance(value, str):
|
|
22
|
+
|
|
23
|
+
def sub(m: re.Match) -> str:
|
|
24
|
+
name, default = m.group(1), m.group(2)
|
|
25
|
+
got = os.environ.get(name)
|
|
26
|
+
if got is None:
|
|
27
|
+
if default is None:
|
|
28
|
+
raise ConfigError(
|
|
29
|
+
f"config references ${{{name}}} but that environment variable is not set"
|
|
30
|
+
)
|
|
31
|
+
return default
|
|
32
|
+
return got
|
|
33
|
+
|
|
34
|
+
return _ENV_RE.sub(sub, value)
|
|
35
|
+
if isinstance(value, dict):
|
|
36
|
+
return {k: interpolate(v) for k, v in value.items()}
|
|
37
|
+
if isinstance(value, list):
|
|
38
|
+
return [interpolate(v) for v in value]
|
|
39
|
+
return value
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class AccountConfig:
|
|
44
|
+
id: str
|
|
45
|
+
#: A label for humans. The id stays the stable handle used in logs and metrics.
|
|
46
|
+
name: str = ""
|
|
47
|
+
type: str = "anthropic_api"
|
|
48
|
+
api_key: str = ""
|
|
49
|
+
#: Override only. Empty means "whatever this account type's adapter defaults to",
|
|
50
|
+
#: so a bedrock or vertex account is not silently pointed at api.anthropic.com.
|
|
51
|
+
base_url: str = ""
|
|
52
|
+
priority: float = 0.0
|
|
53
|
+
#: A routing weight, not a price. `cost_tiered` drains low tiers first and
|
|
54
|
+
#: `sticky_headroom` prefers them on placement, but this number never enters
|
|
55
|
+
#: cost accounting: reported spend and every spend cap come from the model
|
|
56
|
+
#: price table alone. Set it to rank accounts, not to describe a rate.
|
|
57
|
+
cost_tier: float = 1.0
|
|
58
|
+
models: List[str] = field(default_factory=lambda: ["*"])
|
|
59
|
+
max_concurrency: int = 16
|
|
60
|
+
spend_cap_usd: Optional[float] = None
|
|
61
|
+
enabled: bool = True
|
|
62
|
+
headers: Dict[str, str] = field(default_factory=dict)
|
|
63
|
+
#: Provider-specific settings — region, project, model_map, and so on. Kept
|
|
64
|
+
#: untyped so a new adapter needs no change to the config schema.
|
|
65
|
+
options: Dict[str, Any] = field(default_factory=dict)
|
|
66
|
+
#: False when this upstream reports no anthropic-ratelimit-* headers. An
|
|
67
|
+
#: unobservable account must not read as "full", or it wins every comparison
|
|
68
|
+
#: against accounts that honestly report a partly-used budget.
|
|
69
|
+
observable_limits: bool = True
|
|
70
|
+
#: Headroom to assume for an unobservable account. Deliberately middling: it
|
|
71
|
+
#: should neither dominate a healthy pool nor be starved by it.
|
|
72
|
+
assumed_headroom: float = 0.5
|
|
73
|
+
|
|
74
|
+
def supports_model(self, model: str) -> bool:
|
|
75
|
+
for pattern in self.models:
|
|
76
|
+
if pattern == "*" or pattern == model:
|
|
77
|
+
return True
|
|
78
|
+
if pattern.endswith("*") and model.startswith(pattern[:-1]):
|
|
79
|
+
return True
|
|
80
|
+
return False
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class RoutingWeights:
|
|
85
|
+
affinity: float = 0.40
|
|
86
|
+
headroom: float = 0.40
|
|
87
|
+
priority: float = 0.10
|
|
88
|
+
load: float = 0.15
|
|
89
|
+
errors: float = 0.20
|
|
90
|
+
cost: float = 0.10
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class RoutingConfig:
|
|
95
|
+
strategy: str = "sticky_headroom"
|
|
96
|
+
weights: RoutingWeights = field(default_factory=RoutingWeights)
|
|
97
|
+
affinity_ttl_seconds: float = 1800.0
|
|
98
|
+
#: multiplied into every token estimate before it becomes a lease
|
|
99
|
+
estimate_safety_margin: float = 1.15
|
|
100
|
+
|
|
101
|
+
#: How the output half of a lease is sized. `adaptive` predicts from what each
|
|
102
|
+
#: model has actually been returning and is bounded by the caller's own
|
|
103
|
+
#: `max_tokens`, so it can only ever reserve less; `max_tokens` reserves the
|
|
104
|
+
#: ceiling, which is what this did before the estimator existed.
|
|
105
|
+
output_estimate: str = "adaptive"
|
|
106
|
+
output_estimate_quantile: float = 0.95
|
|
107
|
+
output_estimate_min_samples: int = 20
|
|
108
|
+
output_estimate_window: int = 200
|
|
109
|
+
output_estimate_floor: int = 256
|
|
110
|
+
#: Stop predicting for a model once this fraction of its recent completions have
|
|
111
|
+
#: outrun their lease. A p95 predictor is beaten ~5% of the time by design; this
|
|
112
|
+
#: catches a distribution that has changed shape.
|
|
113
|
+
output_estimate_max_undershoot: float = 0.20
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class RetryConfig:
|
|
118
|
+
max_attempts: int = 3
|
|
119
|
+
max_accounts: int = 3
|
|
120
|
+
deadline_seconds: float = 120.0
|
|
121
|
+
backoff_base_seconds: float = 0.25
|
|
122
|
+
backoff_max_seconds: float = 8.0
|
|
123
|
+
#: how long to hold a stream before the first content delta, waiting to see if it fails
|
|
124
|
+
first_token_grace_seconds: float = 20.0
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclass
|
|
128
|
+
class BreakerConfig:
|
|
129
|
+
failure_threshold: int = 5
|
|
130
|
+
cooldown_seconds: float = 30.0
|
|
131
|
+
overloaded_cooldown_seconds: float = 5.0
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass
|
|
135
|
+
class BatchConfig:
|
|
136
|
+
"""Spill lane: batch-priority work can go to the Message Batches API.
|
|
137
|
+
|
|
138
|
+
Off by default. Batches are cheaper but asynchronous, so a request that spills
|
|
139
|
+
holds its connection open while the gateway polls — bounded by the request's own
|
|
140
|
+
wait budget, never longer.
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
enabled: bool = False
|
|
144
|
+
poll_interval_seconds: float = 2.0
|
|
145
|
+
#: Applied to the computed cost of a spilled request. Anthropic prices batch work
|
|
146
|
+
#: below standard; the exact figure is the operator's to supply, like `pricing`.
|
|
147
|
+
cost_multiplier: float = 0.5
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@dataclass
|
|
151
|
+
class QueueConfig:
|
|
152
|
+
max_size: int = 128
|
|
153
|
+
default_max_wait_seconds: float = 60.0
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class ServerConfig:
|
|
158
|
+
host: str = "127.0.0.1"
|
|
159
|
+
port: int = 8787
|
|
160
|
+
allow_remote: bool = False
|
|
161
|
+
request_timeout_seconds: float = 600.0
|
|
162
|
+
#: Re-read the config file when its mtime changes. 0 disables polling; the
|
|
163
|
+
#: /admin/reload endpoint still works either way.
|
|
164
|
+
hot_reload_seconds: float = 5.0
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
@dataclass
|
|
168
|
+
class KeyConfig:
|
|
169
|
+
key: str
|
|
170
|
+
name: str = "default"
|
|
171
|
+
models: List[str] = field(default_factory=lambda: ["*"])
|
|
172
|
+
pool: List[str] = field(default_factory=list)
|
|
173
|
+
rpm: Optional[int] = None
|
|
174
|
+
spend_cap_usd: Optional[float] = None
|
|
175
|
+
#: "interactive" (default) or "batch". Batch traffic yields the queue to
|
|
176
|
+
#: interactive traffic when the pool is saturated.
|
|
177
|
+
priority: str = "interactive"
|
|
178
|
+
#: How long this key's requests will wait for capacity before being told to
|
|
179
|
+
#: come back. Falls back to queue.default_max_wait_seconds.
|
|
180
|
+
max_wait_seconds: Optional[float] = None
|
|
181
|
+
#: Per-conversation caps. `spend_cap_usd` above bounds the whole key; these
|
|
182
|
+
#: bound one conversation within it, which is what catches a runaway agent loop
|
|
183
|
+
#: before it spends the key's entire allowance on a single session.
|
|
184
|
+
session_cap_usd: Optional[float] = None
|
|
185
|
+
session_max_turns: Optional[int] = None
|
|
186
|
+
#: Required to reach /admin/*. A tenant key must not be able to read the pool's
|
|
187
|
+
#: account ids and spend, let alone mint more keys.
|
|
188
|
+
admin: bool = False
|
|
189
|
+
|
|
190
|
+
def supports_model(self, model: str) -> bool:
|
|
191
|
+
for pattern in self.models:
|
|
192
|
+
if pattern == "*" or pattern == model:
|
|
193
|
+
return True
|
|
194
|
+
if pattern.endswith("*") and model.startswith(pattern[:-1]):
|
|
195
|
+
return True
|
|
196
|
+
return False
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
@dataclass
|
|
200
|
+
class ModelPrice:
|
|
201
|
+
"""USD per million tokens.
|
|
202
|
+
|
|
203
|
+
Costs are only reported, and spend caps only enforced, for models priced here.
|
|
204
|
+
Prices come from one of two places, and never from a literal in this module:
|
|
205
|
+
the operator's own `pricing:` block, or the dated `prices.yaml` that ships with
|
|
206
|
+
the release and is opted into with `pricing: builtin`.
|
|
207
|
+
|
|
208
|
+
The rule that produced that split is unchanged — the gateway must never bill
|
|
209
|
+
against a number nobody can attribute. A dated file whose date the console
|
|
210
|
+
displays is attributable; a dict hard-coded here would not be.
|
|
211
|
+
"""
|
|
212
|
+
|
|
213
|
+
input: float = 0.0
|
|
214
|
+
output: float = 0.0
|
|
215
|
+
cache_read: float = 0.0
|
|
216
|
+
cache_write: float = 0.0
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
#: The bundled table, beside this module. Read once, on demand.
|
|
220
|
+
BUILTIN_PRICES_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "prices.yaml")
|
|
221
|
+
|
|
222
|
+
#: Reserved key inside a `pricing:` mapping; no model id collides with it.
|
|
223
|
+
BUILTIN = "builtin"
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def load_builtin_prices() -> Dict[str, Any]:
|
|
227
|
+
"""The shipped price table, or an empty one if it is missing from the install."""
|
|
228
|
+
try:
|
|
229
|
+
with open(BUILTIN_PRICES_PATH, encoding="utf-8") as handle:
|
|
230
|
+
return yaml.safe_load(handle) or {}
|
|
231
|
+
except OSError:
|
|
232
|
+
return {}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@dataclass
|
|
236
|
+
class StoreConfig:
|
|
237
|
+
"""Where shared state lives: affinity, spend, per-key request counts.
|
|
238
|
+
|
|
239
|
+
`memory` keeps everything in-process and loses it on restart. `sqlite` persists
|
|
240
|
+
to one file, which is what makes a spend cap mean anything across a restart.
|
|
241
|
+
`redis` shares state between instances.
|
|
242
|
+
"""
|
|
243
|
+
|
|
244
|
+
backend: str = "memory"
|
|
245
|
+
path: str = "tokenbiryani.db"
|
|
246
|
+
#: Where the encryption key for stored credentials lives. Overridden by the
|
|
247
|
+
#: TOKENBIRYANI_SECRET_KEY environment variable, which containers should use.
|
|
248
|
+
secret_key_path: str = "tokenbiryani.key"
|
|
249
|
+
url: str = "redis://127.0.0.1:6379/0"
|
|
250
|
+
namespace: str = "tokenbiryani"
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
@dataclass
|
|
254
|
+
class SessionConfig:
|
|
255
|
+
"""Per-conversation accounting, and what counts as a runaway."""
|
|
256
|
+
|
|
257
|
+
#: Record per-session cost and turn count on the spend ledger. Two extra ledger
|
|
258
|
+
#: rows per request, which is what per-session caps and the runaway report are
|
|
259
|
+
#: made of. Turn it off and both go quiet rather than lying.
|
|
260
|
+
track: bool = True
|
|
261
|
+
#: Thresholds the runaway report flags at. They enforce nothing on their own —
|
|
262
|
+
#: `keys[].session_cap_usd` and `session_max_turns` do that.
|
|
263
|
+
runaway_turns: int = 200
|
|
264
|
+
runaway_spend_usd: Optional[float] = None
|
|
265
|
+
#: How many sessions the report returns, worst first.
|
|
266
|
+
report_limit: int = 20
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
@dataclass
|
|
270
|
+
class PacingConfig:
|
|
271
|
+
"""Spending a quota window on purpose rather than by accident.
|
|
272
|
+
|
|
273
|
+
Advisory by default: it reports, and changes nothing. `enforcing` throttles
|
|
274
|
+
`batch` priority only — interactive traffic is never delayed to protect a
|
|
275
|
+
budget, because an operator who wants that wants a spend cap, which already
|
|
276
|
+
exists and fails honestly instead of quietly adding latency.
|
|
277
|
+
"""
|
|
278
|
+
|
|
279
|
+
enabled: bool = True
|
|
280
|
+
#: advisory | enforcing
|
|
281
|
+
mode: str = "advisory"
|
|
282
|
+
#: Which reported window to pace against, for accounts that report one: 5h | 7d
|
|
283
|
+
window: str = "7d"
|
|
284
|
+
#: linear | business_hours. A linear target expects a fifth of the quota spent
|
|
285
|
+
#: over a weekend, so a Monday-to-Friday team reads as behind pace every Monday.
|
|
286
|
+
curve: str = "linear"
|
|
287
|
+
ahead_threshold: float = 0.10
|
|
288
|
+
behind_threshold: float = 0.10
|
|
289
|
+
max_batch_delay_seconds: float = 30.0
|
|
290
|
+
#: Required for API-key accounts, which report no weekly window of any kind.
|
|
291
|
+
#: Without it they are simply not paced: inventing a weekly limit would be a
|
|
292
|
+
#: number nobody can attribute.
|
|
293
|
+
weekly_budget_usd: Optional[float] = None
|
|
294
|
+
|
|
295
|
+
#: When enforcing and ahead of pace, send batch-priority work to the Message
|
|
296
|
+
#: Batches API even though the pool has capacity for it now. Batches are priced
|
|
297
|
+
#: below standard and spend a different upstream limit, so this is the cheapest
|
|
298
|
+
#: lever available before anything has to be refused or delayed. Needs
|
|
299
|
+
#: `batch.enabled`; without it there is no lane to prefer and this does nothing.
|
|
300
|
+
prefer_batch_lane_when_ahead: bool = True
|
|
301
|
+
|
|
302
|
+
#: Model substitutions for batch-priority work while ahead of pace, as
|
|
303
|
+
#: {pattern: replacement} with the same trailing-wildcard matching as
|
|
304
|
+
#: `options.model_map`. **Empty by default.** Unlike every other lever here this
|
|
305
|
+
#: changes the answer the caller gets rather than when or where they get it, and
|
|
306
|
+
#: the model is part of the affinity fingerprint, so a conversation that
|
|
307
|
+
#: downshifts mid-flight also takes a cache break. Batch-priority only, never
|
|
308
|
+
#: interactive, and every substitution is recorded on the request.
|
|
309
|
+
model_downshift: Dict[str, str] = field(default_factory=dict)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
@dataclass
|
|
313
|
+
class CacheConfig:
|
|
314
|
+
"""Prompt-cache diagnosis, and the one opt-in that acts on it."""
|
|
315
|
+
|
|
316
|
+
#: Rewrite the caller's body to add a `cache_control` breakpoint at the end of
|
|
317
|
+
#: the stable head when it carries none. Off by default: everywhere else this
|
|
318
|
+
#: gateway routes rather than rewrites, and turning this on makes it the third
|
|
319
|
+
#: exception to that after the two fields Bedrock and Vertex need.
|
|
320
|
+
auto_breakpoint: bool = False
|
|
321
|
+
#: Requests per (key, model) before the advisor will express an opinion.
|
|
322
|
+
advice_min_requests: int = 20
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
@dataclass
|
|
326
|
+
class SpendConfig:
|
|
327
|
+
"""Spend caps are windowed, not lifetime.
|
|
328
|
+
|
|
329
|
+
A lifetime cap on a persistent store would eventually wedge the gateway shut and
|
|
330
|
+
stay that way; a rolling window is what an operator actually means by "cap".
|
|
331
|
+
"""
|
|
332
|
+
|
|
333
|
+
window_hours: float = 24.0
|
|
334
|
+
|
|
335
|
+
@property
|
|
336
|
+
def window_seconds(self) -> float:
|
|
337
|
+
return self.window_hours * 3600.0
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
@dataclass
|
|
341
|
+
class ObservabilityConfig:
|
|
342
|
+
#: prompts are sensitive; never log bodies unless the operator opts in
|
|
343
|
+
log_bodies: bool = False
|
|
344
|
+
event_buffer: int = 500
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
@dataclass
|
|
348
|
+
class OAuthConfig:
|
|
349
|
+
"""Where "Log in with Claude" sends people.
|
|
350
|
+
|
|
351
|
+
Deliberately empty by default. Anthropic does not publish the OAuth endpoints
|
|
352
|
+
its first-party clients use, and this project has never been run against a real
|
|
353
|
+
subscription session, so a hard-coded guess here would look like a working
|
|
354
|
+
feature and fail confusingly. Setting these three is the operator's decision —
|
|
355
|
+
see docs/oauth.md — and until they are set the login flow refuses with an error
|
|
356
|
+
that says exactly what is missing.
|
|
357
|
+
"""
|
|
358
|
+
|
|
359
|
+
client_id: str = ""
|
|
360
|
+
authorize_url: str = ""
|
|
361
|
+
token_url: str = ""
|
|
362
|
+
#: Blank means the manual flow: the provider shows a code and the operator pastes
|
|
363
|
+
#: it into the console. Works without registering a callback anywhere.
|
|
364
|
+
redirect_uri: str = ""
|
|
365
|
+
scopes: List[str] = field(default_factory=list)
|
|
366
|
+
#: How far ahead of expiry a session is renewed, and how often that is checked.
|
|
367
|
+
refresh_skew_seconds: float = 300.0
|
|
368
|
+
refresh_interval_seconds: float = 60.0
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
@dataclass
|
|
372
|
+
class Config:
|
|
373
|
+
server: ServerConfig = field(default_factory=ServerConfig)
|
|
374
|
+
routing: RoutingConfig = field(default_factory=RoutingConfig)
|
|
375
|
+
retry: RetryConfig = field(default_factory=RetryConfig)
|
|
376
|
+
breaker: BreakerConfig = field(default_factory=BreakerConfig)
|
|
377
|
+
queue: QueueConfig = field(default_factory=QueueConfig)
|
|
378
|
+
batch: BatchConfig = field(default_factory=BatchConfig)
|
|
379
|
+
store: StoreConfig = field(default_factory=StoreConfig)
|
|
380
|
+
spend: SpendConfig = field(default_factory=SpendConfig)
|
|
381
|
+
cache: CacheConfig = field(default_factory=CacheConfig)
|
|
382
|
+
pacing: PacingConfig = field(default_factory=PacingConfig)
|
|
383
|
+
sessions: SessionConfig = field(default_factory=SessionConfig)
|
|
384
|
+
observability: ObservabilityConfig = field(default_factory=ObservabilityConfig)
|
|
385
|
+
oauth: OAuthConfig = field(default_factory=OAuthConfig)
|
|
386
|
+
accounts: List[AccountConfig] = field(default_factory=list)
|
|
387
|
+
keys: List[KeyConfig] = field(default_factory=list)
|
|
388
|
+
pricing: Dict[str, ModelPrice] = field(default_factory=dict)
|
|
389
|
+
#: The date on the bundled table, when it is in use. Surfaced by the console so
|
|
390
|
+
#: a stale price is visible rather than silent.
|
|
391
|
+
pricing_as_of: str = ""
|
|
392
|
+
pricing_source: str = ""
|
|
393
|
+
#: Which table the prices above came from: "builtin" (the dated file shipped
|
|
394
|
+
#: with the release), "config" (only what tokenbiryani.yaml names), or "none".
|
|
395
|
+
#: The console can switch this at runtime, which is why the raw block is kept.
|
|
396
|
+
pricing_table: str = "none"
|
|
397
|
+
pricing_raw: Any = None
|
|
398
|
+
path: Optional[str] = None
|
|
399
|
+
|
|
400
|
+
def use_pricing(self, table: str) -> str:
|
|
401
|
+
"""Rebuild the price table from the config file's own `pricing:` block.
|
|
402
|
+
|
|
403
|
+
The operator's entries always win: switching to "builtin" lays the dated
|
|
404
|
+
table underneath them, and switching to "config" takes it away. Neither
|
|
405
|
+
touches the file — a console setting is an overlay, not an edit.
|
|
406
|
+
"""
|
|
407
|
+
raw = self.pricing_raw
|
|
408
|
+
prices: Dict[str, ModelPrice] = {}
|
|
409
|
+
as_of, source = "", ""
|
|
410
|
+
if table == BUILTIN:
|
|
411
|
+
bundled = load_builtin_prices()
|
|
412
|
+
as_of = str(bundled.get("as_of") or "")
|
|
413
|
+
source = str(bundled.get("source") or "")
|
|
414
|
+
for name, spec in (bundled.get("models") or {}).items():
|
|
415
|
+
prices[name] = ModelPrice(**spec)
|
|
416
|
+
if isinstance(raw, dict):
|
|
417
|
+
for name, spec in raw.items():
|
|
418
|
+
if name != BUILTIN:
|
|
419
|
+
prices[name] = ModelPrice(**spec)
|
|
420
|
+
self.pricing = prices
|
|
421
|
+
self.pricing_as_of = as_of
|
|
422
|
+
self.pricing_source = source
|
|
423
|
+
self.pricing_table = table if prices else "none"
|
|
424
|
+
return self.pricing_table
|
|
425
|
+
|
|
426
|
+
def price_for(self, model: str) -> Optional[ModelPrice]:
|
|
427
|
+
if model in self.pricing:
|
|
428
|
+
return self.pricing[model]
|
|
429
|
+
for pattern, price in self.pricing.items():
|
|
430
|
+
if pattern.endswith("*") and model.startswith(pattern[:-1]):
|
|
431
|
+
return price
|
|
432
|
+
return None
|
|
433
|
+
|
|
434
|
+
@classmethod
|
|
435
|
+
def from_dict(cls, raw: Dict[str, Any], path: Optional[str] = None) -> Config:
|
|
436
|
+
raw = interpolate(raw or {})
|
|
437
|
+
|
|
438
|
+
def build(klass, data):
|
|
439
|
+
if not data:
|
|
440
|
+
return klass()
|
|
441
|
+
fields = {f for f in klass.__dataclass_fields__}
|
|
442
|
+
unknown = set(data) - fields
|
|
443
|
+
if unknown:
|
|
444
|
+
raise ConfigError(
|
|
445
|
+
f"unknown {klass.__name__} option(s): {', '.join(sorted(unknown))}"
|
|
446
|
+
)
|
|
447
|
+
return klass(**data)
|
|
448
|
+
|
|
449
|
+
routing_raw = dict(raw.get("routing") or {})
|
|
450
|
+
weights = build(RoutingWeights, routing_raw.pop("weights", None))
|
|
451
|
+
routing = build(RoutingConfig, routing_raw)
|
|
452
|
+
routing.weights = weights
|
|
453
|
+
|
|
454
|
+
# `pricing: builtin` takes the dated table that ships with the release.
|
|
455
|
+
# A mapping may also set `builtin: true` alongside its own entries, which
|
|
456
|
+
# then override it model by model.
|
|
457
|
+
raw_pricing = raw.get("pricing")
|
|
458
|
+
pricing: Dict[str, ModelPrice] = {}
|
|
459
|
+
as_of, source = "", ""
|
|
460
|
+
wants_builtin = raw_pricing == BUILTIN or (
|
|
461
|
+
isinstance(raw_pricing, dict) and bool(raw_pricing.get(BUILTIN))
|
|
462
|
+
)
|
|
463
|
+
if wants_builtin:
|
|
464
|
+
bundled = load_builtin_prices()
|
|
465
|
+
as_of = str(bundled.get("as_of") or "")
|
|
466
|
+
source = str(bundled.get("source") or "")
|
|
467
|
+
pricing.update(
|
|
468
|
+
{name: build(ModelPrice, spec)
|
|
469
|
+
for name, spec in (bundled.get("models") or {}).items()}
|
|
470
|
+
)
|
|
471
|
+
elif isinstance(raw_pricing, str):
|
|
472
|
+
raise ConfigError(
|
|
473
|
+
f"pricing: {raw_pricing!r} is not a thing. Use `pricing: builtin` for the "
|
|
474
|
+
"table that ships with this release, or a mapping of model to prices."
|
|
475
|
+
)
|
|
476
|
+
if isinstance(raw_pricing, dict):
|
|
477
|
+
pricing.update(
|
|
478
|
+
{name: build(ModelPrice, spec)
|
|
479
|
+
for name, spec in raw_pricing.items() if name != BUILTIN}
|
|
480
|
+
)
|
|
481
|
+
accounts = [build(AccountConfig, a) for a in (raw.get("accounts") or [])]
|
|
482
|
+
keys = [build(KeyConfig, k) for k in (raw.get("keys") or [])]
|
|
483
|
+
|
|
484
|
+
seen = set()
|
|
485
|
+
for account in accounts:
|
|
486
|
+
if account.id in seen:
|
|
487
|
+
raise ConfigError(f"duplicate account id: {account.id}")
|
|
488
|
+
seen.add(account.id)
|
|
489
|
+
if account.type == "anthropic_api" and not account.api_key:
|
|
490
|
+
raise ConfigError(f"account {account.id} has no api_key")
|
|
491
|
+
if account.type == "oauth" and account.observable_limits:
|
|
492
|
+
# Subscription sessions report no limits. Left true, the account
|
|
493
|
+
# reads as permanently full and wins every routing comparison.
|
|
494
|
+
account.observable_limits = False
|
|
495
|
+
|
|
496
|
+
for key in keys:
|
|
497
|
+
for account_id in key.pool:
|
|
498
|
+
if account_id not in seen:
|
|
499
|
+
raise ConfigError(
|
|
500
|
+
f"key {key.name} scopes to unknown account {account_id!r}"
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
return cls(
|
|
504
|
+
server=build(ServerConfig, raw.get("server")),
|
|
505
|
+
routing=routing,
|
|
506
|
+
retry=build(RetryConfig, raw.get("retry")),
|
|
507
|
+
breaker=build(BreakerConfig, raw.get("breaker")),
|
|
508
|
+
queue=build(QueueConfig, raw.get("queue")),
|
|
509
|
+
batch=build(BatchConfig, raw.get("batch")),
|
|
510
|
+
store=build(StoreConfig, raw.get("store")),
|
|
511
|
+
spend=build(SpendConfig, raw.get("spend")),
|
|
512
|
+
cache=build(CacheConfig, raw.get("cache")),
|
|
513
|
+
pacing=build(PacingConfig, raw.get("pacing")),
|
|
514
|
+
sessions=build(SessionConfig, raw.get("sessions")),
|
|
515
|
+
observability=build(ObservabilityConfig, raw.get("observability")),
|
|
516
|
+
oauth=build(OAuthConfig, raw.get("oauth")),
|
|
517
|
+
accounts=accounts,
|
|
518
|
+
keys=keys,
|
|
519
|
+
pricing=pricing,
|
|
520
|
+
pricing_as_of=as_of,
|
|
521
|
+
pricing_source=source,
|
|
522
|
+
pricing_table=(
|
|
523
|
+
BUILTIN if wants_builtin else ("config" if pricing else "none")
|
|
524
|
+
),
|
|
525
|
+
pricing_raw=raw_pricing if isinstance(raw_pricing, dict) else None,
|
|
526
|
+
path=path,
|
|
527
|
+
)
|
|
528
|
+
|
|
529
|
+
@classmethod
|
|
530
|
+
def load(cls, path: str) -> Config:
|
|
531
|
+
with open(path, encoding="utf-8") as handle:
|
|
532
|
+
return cls.from_dict(yaml.safe_load(handle) or {}, path=path)
|
|
File without changes
|