gemini-router 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gemini_router-0.1.1/.github/workflows/ci.yml +11 -0
- gemini_router-0.1.1/.gitignore +13 -0
- gemini_router-0.1.1/AGENTS.md +25 -0
- gemini_router-0.1.1/PKG-INFO +67 -0
- gemini_router-0.1.1/README.md +48 -0
- gemini_router-0.1.1/docs/SPEC.md +126 -0
- gemini_router-0.1.1/pyproject.toml +45 -0
- gemini_router-0.1.1/src/gemini_router/__init__.py +23 -0
- gemini_router-0.1.1/src/gemini_router/cli.py +188 -0
- gemini_router-0.1.1/src/gemini_router/config.py +98 -0
- gemini_router-0.1.1/src/gemini_router/errors.py +42 -0
- gemini_router-0.1.1/src/gemini_router/events.py +37 -0
- gemini_router-0.1.1/src/gemini_router/gemini.py +183 -0
- gemini_router-0.1.1/src/gemini_router/ledger.py +223 -0
- gemini_router-0.1.1/src/gemini_router/routing.py +439 -0
- gemini_router-0.1.1/tests/conftest.py +65 -0
- gemini_router-0.1.1/tests/test_cli.py +54 -0
- gemini_router-0.1.1/tests/test_gemini_client.py +138 -0
- gemini_router-0.1.1/tests/test_ledger.py +102 -0
- gemini_router-0.1.1/tests/test_router.py +226 -0
- gemini_router-0.1.1/uv.lock +414 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# AGENTS.md — gemini-router
|
|
2
|
+
|
|
3
|
+
Follow the owner's global AGENTS.md (English in files, Russian in chat, PM-style reports,
|
|
4
|
+
Conventional Commits, SemVer tags `v*`, push after every successful step, docs updated
|
|
5
|
+
with functional changes). Repo-specific rules:
|
|
6
|
+
|
|
7
|
+
## Context
|
|
8
|
+
- Contract: `docs/SPEC.md`; system-wide requirements/architecture/plan live in
|
|
9
|
+
`vernikr/gemini-mcp/docs/01..04`. Keep SPEC.md as-built after every milestone.
|
|
10
|
+
- This library MUST stay dependency-light (`httpx`, `pydantic`, `rich`, stdlib) and
|
|
11
|
+
MCP-agnostic — it is consumed by gemini-mcp, tldr-digest (sync facade) and outsiders.
|
|
12
|
+
- All quota truth = SQLite ledger (`GEMINI_ROUTER_STATE`); no in-memory-only counters.
|
|
13
|
+
Multi-process safety (WAL, busy_timeout) is a hard requirement.
|
|
14
|
+
- Model IDs are volatile (docs/blockers b04 in gemini-mcp): resolve via `listModels`,
|
|
15
|
+
never hardcode availability; every list config-overridable.
|
|
16
|
+
|
|
17
|
+
## Commands (as-built; update me)
|
|
18
|
+
- `uv sync` · `uv run pytest` · `uv run ruff check src tests`
|
|
19
|
+
- `uv run gemini-router quota|models|calls|prune|doctor`
|
|
20
|
+
|
|
21
|
+
## Testing & secrets
|
|
22
|
+
- respx-mocked HTTP only (no network, no real keys in CI/fixtures).
|
|
23
|
+
- Never log or persist raw API keys — `key_id = sha256[:8]`.
|
|
24
|
+
- Ported tldr-digest logic keeps its tested taxonomy (RETRYABLE_STATUS,
|
|
25
|
+
MODEL_SWITCH_STATUS, Retry-After, extract_json) — extend, don't reinvent.
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: gemini-router
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Quota-aware Gemini Flash router: cascading fallback, per-key/model RPD/RPM/TPM ledger (SQLite WAL), key pool, wait policy, CLI
|
|
5
|
+
Author: vernikr
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: gemini,llm,mcp,quota,rate-limit,router
|
|
8
|
+
Requires-Python: >=3.12
|
|
9
|
+
Requires-Dist: httpx>=0.27
|
|
10
|
+
Requires-Dist: pydantic>=2.7
|
|
11
|
+
Requires-Dist: pyyaml>=6.0
|
|
12
|
+
Requires-Dist: rich>=13.7
|
|
13
|
+
Provides-Extra: dev
|
|
14
|
+
Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
|
|
15
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
16
|
+
Requires-Dist: respx>=0.21; extra == 'dev'
|
|
17
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# gemini-router
|
|
21
|
+
|
|
22
|
+
Quota-aware **Gemini Flash router** as a standalone, reusable module — extracted and
|
|
23
|
+
generalized from [`tldr-digest/src/llm.py`](https://github.com/vernikr/tldr-digest).
|
|
24
|
+
|
|
25
|
+
One persistent ledger (SQLite WAL) = the single source of quota truth shared by every
|
|
26
|
+
consumer on a machine: the [gemini-mcp](https://github.com/vernikr/gemini-mcp) gateway,
|
|
27
|
+
`tldr-digest` cron runs, CLI calls, foreign scripts.
|
|
28
|
+
|
|
29
|
+
- Cascading chain with runtime `listModels` validation (default
|
|
30
|
+
`gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`; config-overridable).
|
|
31
|
+
- Per **key×model** budgets: RPD (Pacific-time day + rolling 24 h), RPM, TPM (sliding
|
|
32
|
+
60 s) — pre-flight enforced, accounted from `usageMetadata`; conservative 429 handling
|
|
33
|
+
with cooldowns and `Retry-After`.
|
|
34
|
+
- Key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`), budget-maximizing selection.
|
|
35
|
+
- `thinking_level` per request (Gemini 3.x rules), request cache (0-RPD hits), opt-in
|
|
36
|
+
overflow tiers (`flash-lite → gemma-auto`), event stream for progress frontends.
|
|
37
|
+
- Per-request wait policy for minute-scale gates (RPM/TPM/short cooldowns): default
|
|
38
|
+
**wait** with countdown events (`waiting`, remaining seconds); `wait=False` skips to
|
|
39
|
+
the next candidate or returns wait ETAs in the structured error. Daily resets are
|
|
40
|
+
never waited on.
|
|
41
|
+
- CLI: `gemini-router quota | models | calls | prune | doctor` (rich tables).
|
|
42
|
+
- Deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3` — **no MCP/web-framework coupling**.
|
|
43
|
+
|
|
44
|
+
## Status
|
|
45
|
+
|
|
46
|
+
**v0.1.1 implemented** (milestone M1 of the
|
|
47
|
+
[gemini-mcp plan](https://github.com/vernikr/gemini-mcp/blob/main/docs/04-plan.md)):
|
|
48
|
+
library + CLI, 45 tests green, CI on push. Contract: [`docs/SPEC.md`](docs/SPEC.md).
|
|
49
|
+
|
|
50
|
+
## Planned usage
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from gemini_router import Router, RouterRequest
|
|
54
|
+
router = Router.from_env()
|
|
55
|
+
res = await router.complete(RouterRequest(prompt=..., thinking_level="high"),
|
|
56
|
+
on_event=print)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
|
|
61
|
+
gemini-router quota # RPD left / RPM & TPM windows / cooldowns / next recommendation
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Service-style (no Python): REST `/v1/complete` + `/v1/quota` on the gemini-mcp daemon.
|
|
65
|
+
|
|
66
|
+
Conventions: English in files, Conventional Commits, SemVer tags `v*` (gemini-mcp pins
|
|
67
|
+
a tag), CI = ruff + pytest with mocked HTTP (see `AGENTS.md`).
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# gemini-router
|
|
2
|
+
|
|
3
|
+
Quota-aware **Gemini Flash router** as a standalone, reusable module — extracted and
|
|
4
|
+
generalized from [`tldr-digest/src/llm.py`](https://github.com/vernikr/tldr-digest).
|
|
5
|
+
|
|
6
|
+
One persistent ledger (SQLite WAL) = the single source of quota truth shared by every
|
|
7
|
+
consumer on a machine: the [gemini-mcp](https://github.com/vernikr/gemini-mcp) gateway,
|
|
8
|
+
`tldr-digest` cron runs, CLI calls, foreign scripts.
|
|
9
|
+
|
|
10
|
+
- Cascading chain with runtime `listModels` validation (default
|
|
11
|
+
`gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`; config-overridable).
|
|
12
|
+
- Per **key×model** budgets: RPD (Pacific-time day + rolling 24 h), RPM, TPM (sliding
|
|
13
|
+
60 s) — pre-flight enforced, accounted from `usageMetadata`; conservative 429 handling
|
|
14
|
+
with cooldowns and `Retry-After`.
|
|
15
|
+
- Key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`), budget-maximizing selection.
|
|
16
|
+
- `thinking_level` per request (Gemini 3.x rules), request cache (0-RPD hits), opt-in
|
|
17
|
+
overflow tiers (`flash-lite → gemma-auto`), event stream for progress frontends.
|
|
18
|
+
- Per-request wait policy for minute-scale gates (RPM/TPM/short cooldowns): default
|
|
19
|
+
**wait** with countdown events (`waiting`, remaining seconds); `wait=False` skips to
|
|
20
|
+
the next candidate or returns wait ETAs in the structured error. Daily resets are
|
|
21
|
+
never waited on.
|
|
22
|
+
- CLI: `gemini-router quota | models | calls | prune | doctor` (rich tables).
|
|
23
|
+
- Deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3` — **no MCP/web-framework coupling**.
|
|
24
|
+
|
|
25
|
+
## Status
|
|
26
|
+
|
|
27
|
+
**v0.1.1 implemented** (milestone M1 of the
|
|
28
|
+
[gemini-mcp plan](https://github.com/vernikr/gemini-mcp/blob/main/docs/04-plan.md)):
|
|
29
|
+
library + CLI, 45 tests green, CI on push. Contract: [`docs/SPEC.md`](docs/SPEC.md).
|
|
30
|
+
|
|
31
|
+
## Planned usage
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from gemini_router import Router, RouterRequest
|
|
35
|
+
router = Router.from_env()
|
|
36
|
+
res = await router.complete(RouterRequest(prompt=..., thinking_level="high"),
|
|
37
|
+
on_event=print)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
|
|
42
|
+
gemini-router quota # RPD left / RPM & TPM windows / cooldowns / next recommendation
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Service-style (no Python): REST `/v1/complete` + `/v1/quota` on the gemini-mcp daemon.
|
|
46
|
+
|
|
47
|
+
Conventions: English in files, Conventional Commits, SemVer tags `v*` (gemini-mcp pins
|
|
48
|
+
a tag), CI = ruff + pytest with mocked HTTP (see `AGENTS.md`).
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# gemini-router · SPEC (stage 2/3 artifact slice)
|
|
2
|
+
|
|
3
|
+
Date: 2026-10-08 · Companion docs live in `vernikr/gemini-mcp/docs/01..04` (system-wide
|
|
4
|
+
requirements, architecture, plan). This file is the router's own contract; updated as-built
|
|
5
|
+
each milestone.
|
|
6
|
+
|
|
7
|
+
## 1. What it is
|
|
8
|
+
|
|
9
|
+
Quota-aware Gemini Flash router extracted from `tldr-digest/src/llm.py`, generalized:
|
|
10
|
+
|
|
11
|
+
- cascading model chain with runtime `listModels` validation (default
|
|
12
|
+
`gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`);
|
|
13
|
+
- per **key×model** budgets: RPD (America/Los_Angeles day + rolling 24 h), RPM and TPM
|
|
14
|
+
(sliding 60 s) — enforced pre-flight, accounted post-fact from `usageMetadata`;
|
|
15
|
+
- key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`); selection prefers the pair with
|
|
16
|
+
the largest remaining budget;
|
|
17
|
+
- persistent multi-process ledger (SQLite WAL) — the single source of quota truth shared
|
|
18
|
+
by every consumer on the machine (MCP daemon, stdio sessions, cron jobs, foreign scripts);
|
|
19
|
+
- 429/5xx/4xx taxonomy inherited from tldr-digest (`RETRYABLE_STATUS`,
|
|
20
|
+
`MODEL_SWITCH_STATUS`, `Retry-After`, backoff+jitter, provider deadline);
|
|
21
|
+
- opt-in overflow tiers (`gemini-3.5-flash-lite`, best `gemma-*`) — never automatic;
|
|
22
|
+
- request cache (hash of model+params+prompt, TTL) — hits spend 0 RPD;
|
|
23
|
+
- event stream (`on_event`) for progress reporting by frontends;
|
|
24
|
+
- `thinking_level` (low|medium|high) per request; 3.x rules applied (no
|
|
25
|
+
temperature/top_p for Gemini 3.x; `thinking_level` not `thinking_budget`);
|
|
26
|
+
- CLI: `quota`, `models`, `calls`, `prune`, `doctor`; sync facade for legacy callers.
|
|
27
|
+
|
|
28
|
+
Zero MCP/web-framework deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3`. Python ≥3.12.
|
|
29
|
+
|
|
30
|
+
## 2. Public API (v1 contract)
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from gemini_router import Router, RouterRequest, RouterResult, RouterEvent, QuotaExhaustedError
|
|
34
|
+
|
|
35
|
+
router = Router.from_env() # env + optional GEMINI_ROUTER_CONFIG yaml
|
|
36
|
+
res: RouterResult = await router.complete(
|
|
37
|
+
RouterRequest(
|
|
38
|
+
prompt=str, system=str | None = None,
|
|
39
|
+
thinking_level="low|medium|high" = "high",
|
|
40
|
+
output_format="text|json" = "text",
|
|
41
|
+
max_output_tokens=8192,
|
|
42
|
+
kind="single|chunk|consolidate|count" = "single",
|
|
43
|
+
request_id=None, source="cli", overflow="off|lite|lite+gemma",
|
|
44
|
+
wait=True, # wait out minute-scale gates (RPM/TPM/short cooldown,
|
|
45
|
+
# ≤ wait.max_s), countdown via `waiting` events;
|
|
46
|
+
# daily RPD resets are NEVER waited on
|
|
47
|
+
model=None), # pin → skips chain fallback
|
|
48
|
+
on_event=callable | None) # RouterEvent(stage, pct, model, key_id, wait_s, detail, ts)
|
|
49
|
+
# stages: routing|waiting|dispatch|retry|switch|overflow|ok|quota_exhausted
|
|
50
|
+
|
|
51
|
+
res.text, res.model, res.key_id, res.tokens_in, res.tokens_out
|
|
52
|
+
res.latency_s, res.cached, res.overflow_tier, res.events, res.quota_after
|
|
53
|
+
await router.count_tokens(text, model=None) -> int
|
|
54
|
+
await router.quota_table() -> list[QuotaRow] # key_id, model, rpd_left, rpm_wait_s,
|
|
55
|
+
# tpm_wait_s, cooldown_until, recommended
|
|
56
|
+
router.complete_sync(...) # blocking facade (tldr-digest migration)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Errors: `QuotaExhaustedError` carries the machine-readable payload
|
|
60
|
+
(per-model rpd_left/rpm_wait_s/tpm_wait_s/cooldown_until, reset_eta_pt, hint) — frontends
|
|
61
|
+
surface it as-is. `RouterError` base for everything else (network, http 4xx, parse).
|
|
62
|
+
|
|
63
|
+
## 3. Ledger (SQLite, `GEMINI_ROUTER_STATE`, default `/opt/apps/shared/gemini-router/state.db`)
|
|
64
|
+
|
|
65
|
+
Tables `calls`, `cooldowns`, `cache`, `meta` — full DDL and accounting rules:
|
|
66
|
+
gemini-mcp `docs/03-architecture.md` §3 (single canonical definition).
|
|
67
|
+
Rules summary: WAL + busy_timeout 5 s; TPM pre-flight writes a `status='planned'` row,
|
|
68
|
+
upgraded after the call; RPD counts `ok|quota|http5xx` attempts (conservative),
|
|
69
|
+
not client-validation 4xx/network; PT-midnight + rolling-24h dual gate; retention prune
|
|
70
|
+
via `gemini-router prune` and by frontends' daily task.
|
|
71
|
+
|
|
72
|
+
## 4. Config
|
|
73
|
+
|
|
74
|
+
Env: `GEMINI_API_KEYS`/`GEMINI_API_KEY_1..10`, `GEMINI_ROUTER_STATE`,
|
|
75
|
+
`GEMINI_ROUTER_CONFIG` (yaml). YAML keys: `chain`, `overflow_tiers`,
|
|
76
|
+
`limits_per_key_model {rpd,rpm,tpm}`, `overflow_limits {rpd,rpm,tpm}`, `limits_overrides`,
|
|
77
|
+
`quota_day_tz`, `retry {max,backoff_base_s,jitter_max_s,provider_timeout_s,request_delay_s}`,
|
|
78
|
+
`wait {enabled_default,max_s,poll_s}` (FR-C13 semantics), `window_s`, `planned_ttl_s`,
|
|
79
|
+
`models_refresh_h`, `cache_ttl_h`, `state_path`, `base_url`, `temperature` (pre-3.x models
|
|
80
|
+
only), `max_output_tokens_default`, `chars_per_token`. Defaults per gemini-mcp `docs/03-architecture.md` §8. Secrets never logged
|
|
81
|
+
(`key_id = sha256(key)[:8]`).
|
|
82
|
+
|
|
83
|
+
## 5. Integration guide (external consumers)
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
|
|
87
|
+
# private repo: export UV_INDEX / use gh auth or a deploy key; see README
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Service-style consumption (no Python): REST `POST /v1/complete`, `GET /v1/quota` on the
|
|
91
|
+
gemini-mcp daemon (localhost:8790, bearer) — same ledger, same rules.
|
|
92
|
+
|
|
93
|
+
## 6. Extraction mapping from tldr-digest/src/llm.py
|
|
94
|
+
|
|
95
|
+
| tldr-digest | router | Change |
|
|
96
|
+
|---|---|---|
|
|
97
|
+
| `DEFAULT_MODEL_PRIORITY`/`llm_model_priority` | `config.chain` | Gemini-only in v1; groq/openrouter stay in tldr until a provider abstraction is justified |
|
|
98
|
+
| `_resolve_models`/`_list_models`/`_best_available` | `routing.resolve_chain` | + refresh interval, + preview-slot heuristics |
|
|
99
|
+
| `RETRYABLE_STATUS`/`MODEL_SWITCH_STATUS`/`_parse_retry_after`/backoff | `routing` | ported as-is |
|
|
100
|
+
| `Provider.can_call/register_call` (in-memory) | `ledger` queries | persistent, per key×model, + TPM |
|
|
101
|
+
| `_call_gemini` | `gemini.generate` | async httpx, thinking_level, usageMetadata accounting |
|
|
102
|
+
| `extract_json` | `gemini.extract_json` | ported as-is |
|
|
103
|
+
| `complete_json` | `complete(output_format="json")` | superset |
|
|
104
|
+
|
|
105
|
+
## 7. Versioning & CI
|
|
106
|
+
|
|
107
|
+
SemVer; tags `v*`; gemini-mcp pins a tag. CI: ruff + pytest (respx mocks, no network) on
|
|
108
|
+
push/PR (free runners). Conventional Commits.
|
|
109
|
+
|
|
110
|
+
## 8. As-built notes
|
|
111
|
+
|
|
112
|
+
- v0.1.1 (2026-10-08, live VPS smoke): **thinking-token starvation** discovered —
|
|
113
|
+
in 3.x models `maxOutputTokens` counts thinking tokens, so a tiny cap (16) yields
|
|
114
|
+
empty visible text and no `candidatesTokenCount`. Fixes: thought parts
|
|
115
|
+
(`thought=true`) never join visible text; `GenResult`/`RouterResult` carry
|
|
116
|
+
`finish_reason` + `tokens_thought`; `max_output_tokens_default` 8192→32768
|
|
117
|
+
(cap only, no spend implication). 45 tests green.
|
|
118
|
+
- v0.1.0 (M1): full contract above implemented; 43 tests green (ledger dual-gate RPD,
|
|
119
|
+
planned-row TPM windows, wait policy incl. cap and countdown events, cache +
|
|
120
|
+
request_id idempotency, overflow lite→gemma, live 429-day cooldown, multi-key pool,
|
|
121
|
+
countTokens spends no RPD, pinned-model handling).
|
|
122
|
+
- `extract_json` improvement over the tldr original: when several JSON candidates match,
|
|
123
|
+
the earliest starting in the text wins (fixes nested `{…[…]…}` extraction).
|
|
124
|
+
- Ledger statuses: `planned|ok|quota|http4xx|http5xx|network|blocked|skipped`;
|
|
125
|
+
RPD-conservative set = `ok,quota,http5xx,blocked` (D3).
|
|
126
|
+
- CLI exit codes: 0 ok, 1 doctor failed, 2 config/usage error.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "gemini-router"
|
|
3
|
+
version = "0.1.1"
|
|
4
|
+
description = "Quota-aware Gemini Flash router: cascading fallback, per-key/model RPD/RPM/TPM ledger (SQLite WAL), key pool, wait policy, CLI"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.12"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
authors = [{ name = "vernikr" }]
|
|
9
|
+
keywords = ["gemini", "router", "quota", "rate-limit", "llm", "mcp"]
|
|
10
|
+
dependencies = [
|
|
11
|
+
"httpx>=0.27",
|
|
12
|
+
"pydantic>=2.7",
|
|
13
|
+
"pyyaml>=6.0",
|
|
14
|
+
"rich>=13.7",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.optional-dependencies]
|
|
18
|
+
dev = [
|
|
19
|
+
"pytest>=8.0",
|
|
20
|
+
"pytest-asyncio>=0.24",
|
|
21
|
+
"respx>=0.21",
|
|
22
|
+
"ruff>=0.6",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
[project.scripts]
|
|
26
|
+
gemini-router = "gemini_router.cli:main"
|
|
27
|
+
|
|
28
|
+
[build-system]
|
|
29
|
+
requires = ["hatchling"]
|
|
30
|
+
build-backend = "hatchling.build"
|
|
31
|
+
|
|
32
|
+
[tool.hatch.build.targets.wheel]
|
|
33
|
+
packages = ["src/gemini_router"]
|
|
34
|
+
|
|
35
|
+
[tool.pytest.ini_options]
|
|
36
|
+
asyncio_mode = "auto"
|
|
37
|
+
testpaths = ["tests"]
|
|
38
|
+
|
|
39
|
+
[tool.ruff]
|
|
40
|
+
line-length = 100
|
|
41
|
+
target-version = "py312"
|
|
42
|
+
src = ["src", "tests"]
|
|
43
|
+
|
|
44
|
+
[tool.ruff.lint]
|
|
45
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""gemini-router — quota-aware Gemini Flash router with a persistent ledger.
|
|
2
|
+
|
|
3
|
+
Public API (docs/SPEC.md §2):
|
|
4
|
+
Router, RouterRequest, RouterResult, RouterEvent,
|
|
5
|
+
RouterError, GeminiHTTPError, QuotaExhaustedError,
|
|
6
|
+
Config, Limits, Ledger, GeminiClient, extract_json, key_id, estimate_tokens
|
|
7
|
+
"""
|
|
8
|
+
from .config import Config, Limits, RetryCfg, WaitCfg, estimate_tokens, key_id
|
|
9
|
+
from .errors import GeminiHTTPError, QuotaExhaustedError, RouterError
|
|
10
|
+
from .events import RouterEvent
|
|
11
|
+
from .gemini import GeminiClient, GenResult, extract_json, is_gemini3
|
|
12
|
+
from .ledger import Ledger, iso, pt_midnight_ts
|
|
13
|
+
from .routing import Router, RouterRequest, RouterResult, cache_hash, gates_for, pick_gemma
|
|
14
|
+
|
|
15
|
+
__version__ = "0.1.1"
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"Router", "RouterRequest", "RouterResult", "RouterEvent",
|
|
19
|
+
"RouterError", "GeminiHTTPError", "QuotaExhaustedError",
|
|
20
|
+
"Config", "Limits", "RetryCfg", "WaitCfg", "Ledger", "GeminiClient", "GenResult",
|
|
21
|
+
"extract_json", "is_gemini3", "key_id", "estimate_tokens", "cache_hash",
|
|
22
|
+
"gates_for", "pick_gemma", "pt_midnight_ts", "iso", "__version__",
|
|
23
|
+
]
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""Terminal UX (FR-C12): quota | models | calls | prune | doctor. Works over SSH
|
|
2
|
+
against the shared state file without touching the daemon (FR-E7)."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import asyncio
|
|
7
|
+
import json
|
|
8
|
+
|
|
9
|
+
from rich.console import Console
|
|
10
|
+
from rich.table import Table
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .config import Config
|
|
14
|
+
from .errors import RouterError
|
|
15
|
+
from .ledger import Ledger, iso, pt_midnight_ts
|
|
16
|
+
from .routing import Router, pick_gemma
|
|
17
|
+
|
|
18
|
+
console = Console()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _load_cfg(args) -> Config:
|
|
22
|
+
over = {"state_path": args.state} if getattr(args, "state", None) else {}
|
|
23
|
+
return Config.load(getattr(args, "config", None), **over)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _router(cfg: Config) -> Router:
|
|
27
|
+
return Router(cfg, ledger=Ledger(cfg.state_path))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def cmd_quota(cfg: Config, args) -> int:
|
|
31
|
+
rows = _router(cfg).quota_table()
|
|
32
|
+
if args.json:
|
|
33
|
+
print(json.dumps(rows, indent=1, default=str))
|
|
34
|
+
return 0
|
|
35
|
+
t = Table(title="Gemini quota (per key×model)", title_style="bold cyan", expand=True)
|
|
36
|
+
for col in ("key", "model", "RPD left", "RPM win", "TPM win", "cooldown", "next"):
|
|
37
|
+
t.add_column(col)
|
|
38
|
+
for r in rows:
|
|
39
|
+
wait = f"{r['wait_s']:.0f}s ({r['reason']})" if r["wait_s"] > 0 else "—"
|
|
40
|
+
style = "green" if r["rpd_left"] > 0 and r["wait_s"] <= 0 else (
|
|
41
|
+
"yellow" if r["rpd_left"] > 0 else "red")
|
|
42
|
+
rpm_w = wait if r["reason"] == "rpm" else ""
|
|
43
|
+
tpm_w = wait if r["reason"] == "tpm" else ""
|
|
44
|
+
t.add_row(r["key_id"], f"[{style}]{r['model']}[/{style}]",
|
|
45
|
+
f"{r['rpd_left']}/{r['rpd_limit']}",
|
|
46
|
+
f"{r['rpm_used']}/{r['rpm_limit']} {rpm_w}".strip(),
|
|
47
|
+
f"{r['tpm_used']}/{r['tpm_limit']} {tpm_w}".strip(),
|
|
48
|
+
iso(r["cooldown_until"])[:19] if r["cooldown_until"] else "—",
|
|
49
|
+
"★" if r["recommended"] else "")
|
|
50
|
+
console.print(t)
|
|
51
|
+
return 0
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def cmd_models(cfg: Config, args) -> int:
|
|
55
|
+
async def _run() -> dict:
|
|
56
|
+
r = _router(cfg)
|
|
57
|
+
try:
|
|
58
|
+
return await r.availability() if not args.offline else {}
|
|
59
|
+
finally:
|
|
60
|
+
await r.aclose()
|
|
61
|
+
avail = asyncio.run(_run())
|
|
62
|
+
union: set[str] = set()
|
|
63
|
+
for s in avail.values():
|
|
64
|
+
if s:
|
|
65
|
+
union |= s
|
|
66
|
+
out = {"chain": cfg.chain, "overflow_tiers": cfg.overflow_tiers,
|
|
67
|
+
"gemma_pick": pick_gemma(union) if union else None,
|
|
68
|
+
"per_key": {k: (sorted(v) if v else "listModels unavailable → trust config")
|
|
69
|
+
for k, v in avail.items()},
|
|
70
|
+
"missing_from_live": sorted(m for m in cfg.chain if union and m not in union)}
|
|
71
|
+
if args.json:
|
|
72
|
+
print(json.dumps(out, indent=1))
|
|
73
|
+
return 0
|
|
74
|
+
console.print_json(json.dumps(out, indent=1))
|
|
75
|
+
return 0
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def cmd_calls(cfg: Config, args) -> int:
|
|
79
|
+
rows = Ledger(cfg.state_path).calls_tail(
|
|
80
|
+
args.n, model=args.model, status=args.status, source=args.source,
|
|
81
|
+
request_id=args.request_id)
|
|
82
|
+
if args.json:
|
|
83
|
+
print(json.dumps(rows, indent=1, default=str))
|
|
84
|
+
return 0
|
|
85
|
+
t = Table(title=f"Last {len(rows)} calls", expand=True)
|
|
86
|
+
for col in ("ts (UTC)", "src", "kind", "key", "model", "status", "http",
|
|
87
|
+
"in", "out", "ms", "tier"):
|
|
88
|
+
t.add_column(col, overflow="fold")
|
|
89
|
+
for r in reversed(rows):
|
|
90
|
+
style = {"ok": "green", "quota": "red", "http5xx": "yellow",
|
|
91
|
+
"network": "yellow", "blocked": "magenta"}.get(r["status"], "dim")
|
|
92
|
+
t.add_row(r["ts_iso"][:19], r["source"], r["kind"], r["key_id"], r["model"],
|
|
93
|
+
f"[{style}]{r['status']}[/{style}]", str(r["http_status"] or ""),
|
|
94
|
+
str(r["tokens_in"] or ""), str(r["tokens_out"] or ""),
|
|
95
|
+
str(r["latency_ms"] or ""), str(r["overflow_tier"]))
|
|
96
|
+
console.print(t)
|
|
97
|
+
return 0
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def cmd_prune(cfg: Config, args) -> int:
|
|
101
|
+
res = Ledger(cfg.state_path).prune(
|
|
102
|
+
audit_keep_days=args.audit_days, cache_ttl_s=cfg.cache_ttl_h * 3600, dry=args.dry)
|
|
103
|
+
console.print(("DRY " if args.dry else "") + json.dumps(res))
|
|
104
|
+
return 0
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def cmd_doctor(cfg: Config, args) -> int:
|
|
108
|
+
ok = True
|
|
109
|
+
console.print(f"[bold]gemini-router {__version__} doctor[/bold]")
|
|
110
|
+
console.print(f" state: {cfg.state_path}")
|
|
111
|
+
try:
|
|
112
|
+
led = Ledger(cfg.state_path)
|
|
113
|
+
console.print(f" state stats: {led.stats()}")
|
|
114
|
+
except Exception as e: # noqa: BLE001
|
|
115
|
+
console.print(f" [red]state unusable: {e}[/red]")
|
|
116
|
+
return 1
|
|
117
|
+
if cfg.api_keys:
|
|
118
|
+
from .config import key_id
|
|
119
|
+
console.print(f" keys: {len(cfg.api_keys)} → "
|
|
120
|
+
+ ", ".join(key_id(k) for k in cfg.api_keys))
|
|
121
|
+
else:
|
|
122
|
+
console.print(" [red]no API keys configured "
|
|
123
|
+
"(GEMINI_API_KEYS / GEMINI_API_KEY_1..10)[/red]")
|
|
124
|
+
ok = False
|
|
125
|
+
tz_now = pt_midnight_ts(cfg.quota_day_tz)
|
|
126
|
+
console.print(f" quota day ({cfg.quota_day_tz}): resets in "
|
|
127
|
+
f"{(pt_midnight_ts(cfg.quota_day_tz, days_ahead=1) - tz_now) / 3600:.1f} h "
|
|
128
|
+
f"(midnight was {iso(tz_now)[:19]} UTC)")
|
|
129
|
+
console.print(f" chain: {' → '.join(cfg.chain)}")
|
|
130
|
+
console.print(f" limits/key×model: {cfg.limits_per_key_model.model_dump()} · "
|
|
131
|
+
f"overflow: {cfg.overflow_limits.model_dump()} · "
|
|
132
|
+
f"wait: {cfg.wait.model_dump()}")
|
|
133
|
+
if not args.offline and cfg.api_keys:
|
|
134
|
+
async def _probe() -> None:
|
|
135
|
+
r = _router(cfg)
|
|
136
|
+
try:
|
|
137
|
+
avail = await r.availability()
|
|
138
|
+
for kid, models in avail.items():
|
|
139
|
+
if models is None:
|
|
140
|
+
console.print(f" [yellow]key {kid}: listModels failed (offline?)[/yellow]")
|
|
141
|
+
else:
|
|
142
|
+
have = [m for m in cfg.chain if m in models]
|
|
143
|
+
console.print(f" key {kid}: {len(models)} models live; chain available: "
|
|
144
|
+
+ (", ".join(have) or "[red]NONE[/red]"))
|
|
145
|
+
finally:
|
|
146
|
+
await r.aclose()
|
|
147
|
+
asyncio.run(_probe())
|
|
148
|
+
console.print("[green]OK[/green]" if ok else "[red]FAILED[/red]")
|
|
149
|
+
return 0 if ok else 1
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def main(argv: list[str] | None = None) -> int:
|
|
153
|
+
ap = argparse.ArgumentParser(prog="gemini-router",
|
|
154
|
+
description="Quota-aware Gemini Flash router CLI")
|
|
155
|
+
ap.add_argument("--config", help="YAML config path (or GEMINI_ROUTER_CONFIG)")
|
|
156
|
+
ap.add_argument("--state", help="override state db path (or GEMINI_ROUTER_STATE)")
|
|
157
|
+
ap.add_argument("--version", action="version", version=f"gemini-router {__version__}")
|
|
158
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
159
|
+
|
|
160
|
+
p = sub.add_parser("quota", help="live budget table (which model can take a request)")
|
|
161
|
+
p.add_argument("--json", action="store_true")
|
|
162
|
+
p = sub.add_parser("models", help="resolved chain vs live listModels")
|
|
163
|
+
p.add_argument("--offline", action="store_true")
|
|
164
|
+
p.add_argument("--json", action="store_true")
|
|
165
|
+
p = sub.add_parser("calls", help="audit log tail")
|
|
166
|
+
p.add_argument("-n", type=int, default=50)
|
|
167
|
+
for flag in ("--model", "--status", "--source", "--request-id"):
|
|
168
|
+
p.add_argument(flag)
|
|
169
|
+
p.add_argument("--json", action="store_true")
|
|
170
|
+
p = sub.add_parser("prune", help="apply retention (audit days, cache TTL)")
|
|
171
|
+
p.add_argument("--audit-days", type=float, default=30)
|
|
172
|
+
p.add_argument("--dry", action="store_true")
|
|
173
|
+
p = sub.add_parser("doctor", help="config/state/connectivity check")
|
|
174
|
+
p.add_argument("--offline", action="store_true")
|
|
175
|
+
|
|
176
|
+
args = ap.parse_args(argv)
|
|
177
|
+
handlers = {"quota": cmd_quota, "models": cmd_models, "calls": cmd_calls,
|
|
178
|
+
"prune": cmd_prune, "doctor": cmd_doctor}
|
|
179
|
+
try:
|
|
180
|
+
cfg = _load_cfg(args)
|
|
181
|
+
return handlers[args.cmd](cfg, args)
|
|
182
|
+
except RouterError as e:
|
|
183
|
+
console.print(f"[red]error:[/red] {e}")
|
|
184
|
+
return 2
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
if __name__ == "__main__":
|
|
188
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Config: pydantic model + YAML/env loading (SPEC §4). Defaults mirror
|
|
2
|
+
gemini-mcp docs/03 §8. Secrets come only from env; never from YAML."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import yaml
|
|
10
|
+
from pydantic import BaseModel, Field
|
|
11
|
+
|
|
12
|
+
DEFAULT_CHAIN = ["gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash",
|
|
13
|
+
"gemini-3.5-flash", "gemini-3-flash-preview"]
|
|
14
|
+
DEFAULT_OVERFLOW = ["gemini-3.5-flash-lite", "gemma-auto"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Limits(BaseModel):
|
|
18
|
+
rpd: int = 20
|
|
19
|
+
rpm: int = 5
|
|
20
|
+
tpm: int = 250_000
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class RetryCfg(BaseModel):
|
|
24
|
+
max: int = 3
|
|
25
|
+
backoff_base_s: float = 5.0
|
|
26
|
+
jitter_max_s: float = 2.0
|
|
27
|
+
provider_timeout_s: float = 120.0
|
|
28
|
+
request_delay_s: float = 2.0
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class WaitCfg(BaseModel):
|
|
32
|
+
enabled_default: bool = True # FR-C13: wait out minute-scale gates by default
|
|
33
|
+
max_s: float = 65.0
|
|
34
|
+
poll_s: float = 2.0
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class Config(BaseModel):
|
|
38
|
+
chain: list[str] = Field(default_factory=lambda: list(DEFAULT_CHAIN))
|
|
39
|
+
overflow_tiers: list[str] = Field(default_factory=lambda: list(DEFAULT_OVERFLOW))
|
|
40
|
+
limits_per_key_model: Limits = Field(default_factory=Limits)
|
|
41
|
+
overflow_limits: Limits = Field(default_factory=lambda: Limits(rpd=500, rpm=15, tpm=250_000))
|
|
42
|
+
limits_overrides: dict[str, dict[str, int]] = Field(default_factory=dict)
|
|
43
|
+
quota_day_tz: str = "America/Los_Angeles"
|
|
44
|
+
retry: RetryCfg = Field(default_factory=RetryCfg)
|
|
45
|
+
wait: WaitCfg = Field(default_factory=WaitCfg)
|
|
46
|
+
window_s: float = 60.0 # sliding RPM/TPM window
|
|
47
|
+
planned_ttl_s: float = 300.0 # stale 'planned' rows stop counting after this
|
|
48
|
+
models_refresh_h: float = 12.0
|
|
49
|
+
cache_ttl_h: float = 24.0
|
|
50
|
+
state_path: str = "state.db"
|
|
51
|
+
base_url: str = "https://generativelanguage.googleapis.com/v1beta"
|
|
52
|
+
api_keys: list[str] = Field(default_factory=list)
|
|
53
|
+
temperature: float | None = None # applied to pre-3.x models only
|
|
54
|
+
max_output_tokens_default: int = 32768 # cap only; 3.x thinking tokens count toward it
|
|
55
|
+
chars_per_token: int = 4 # estimate when countTokens not provided
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def load(cls, path: str | Path | None = None, **over) -> Config:
|
|
59
|
+
data: dict = {}
|
|
60
|
+
p = path or os.environ.get("GEMINI_ROUTER_CONFIG")
|
|
61
|
+
if p and Path(p).exists():
|
|
62
|
+
raw = yaml.safe_load(Path(p).read_text(encoding="utf-8")) or {}
|
|
63
|
+
data = raw.get("router", raw) if isinstance(raw.get("router"), dict) else raw
|
|
64
|
+
keys = _env_keys()
|
|
65
|
+
if keys:
|
|
66
|
+
data["api_keys"] = keys
|
|
67
|
+
if os.environ.get("GEMINI_ROUTER_STATE"):
|
|
68
|
+
data["state_path"] = os.environ["GEMINI_ROUTER_STATE"]
|
|
69
|
+
data.update({k: v for k, v in over.items() if v is not None})
|
|
70
|
+
return cls.model_validate(data)
|
|
71
|
+
|
|
72
|
+
def limits_for(self, model: str) -> Limits:
|
|
73
|
+
base = (self.limits_per_key_model if model in self.chain else self.overflow_limits)
|
|
74
|
+
merged = base.model_dump() | self.limits_overrides.get(model, {})
|
|
75
|
+
return Limits(**merged)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _env_keys() -> list[str]:
|
|
79
|
+
keys: list[str] = []
|
|
80
|
+
v = os.environ.get("GEMINI_API_KEYS", "")
|
|
81
|
+
keys += [k.strip() for k in v.split(",") if k.strip()]
|
|
82
|
+
for i in range(1, 11):
|
|
83
|
+
k = os.environ.get(f"GEMINI_API_KEY_{i}", "").strip()
|
|
84
|
+
if k:
|
|
85
|
+
keys.append(k)
|
|
86
|
+
k = os.environ.get("GEMINI_API_KEY", "").strip()
|
|
87
|
+
if k:
|
|
88
|
+
keys.append(k)
|
|
89
|
+
return list(dict.fromkeys(keys))
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def key_id(key: str) -> str:
|
|
93
|
+
"""Masked key identity for logs/ledger (FR-E8): never store raw keys."""
|
|
94
|
+
return hashlib.sha256(key.encode()).hexdigest()[:8]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def estimate_tokens(text: str, chars_per_token: int = 4) -> int:
|
|
98
|
+
return max(1, len(text) // max(1, chars_per_token))
|