gemini-router 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,11 @@
1
+ name: ci
2
+ on: [push, pull_request]
3
+ jobs:
4
+ test:
5
+ runs-on: ubuntu-latest
6
+ steps:
7
+ - uses: actions/checkout@v4
8
+ - uses: astral-sh/setup-uv@v5
9
+ - run: uv sync --extra dev
10
+ - run: uv run ruff check src tests
11
+ - run: uv run pytest -q
@@ -0,0 +1,13 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ .env
5
+ *.db
6
+ *.db-wal
7
+ *.db-shm
8
+ logs/
9
+ .pytest_cache/
10
+ .ruff_cache/
11
+ dist/
12
+ *.egg-info/
13
+ .DS_Store
@@ -0,0 +1,25 @@
1
+ # AGENTS.md — gemini-router
2
+
3
+ Follow the owner's global AGENTS.md (English in files, Russian in chat, PM-style reports,
4
+ Conventional Commits, SemVer tags `v*`, push after every successful step, docs updated
5
+ with functional changes). Repo-specific rules:
6
+
7
+ ## Context
8
+ - Contract: `docs/SPEC.md`; system-wide requirements/architecture/plan live in
9
+ `vernikr/gemini-mcp/docs/01..04`. Keep SPEC.md as-built after every milestone.
10
+ - This library MUST stay dependency-light (`httpx`, `pydantic`, `rich`, stdlib) and
11
+ MCP-agnostic — it is consumed by gemini-mcp, tldr-digest (sync facade) and outsiders.
12
+ - All quota truth = SQLite ledger (`GEMINI_ROUTER_STATE`); no in-memory-only counters.
13
+ Multi-process safety (WAL, busy_timeout) is a hard requirement.
14
+ - Model IDs are volatile (docs/blockers b04 in gemini-mcp): resolve via `listModels`,
15
+ never hardcode availability; every list config-overridable.
16
+
17
+ ## Commands (as-built; update me)
18
+ - `uv sync` · `uv run pytest` · `uv run ruff check src tests`
19
+ - `uv run gemini-router quota|models|calls|prune|doctor`
20
+
21
+ ## Testing & secrets
22
+ - respx-mocked HTTP only (no network, no real keys in CI/fixtures).
23
+ - Never log or persist raw API keys — `key_id = sha256[:8]`.
24
+ - Ported tldr-digest logic keeps its tested taxonomy (RETRYABLE_STATUS,
25
+ MODEL_SWITCH_STATUS, Retry-After, extract_json) — extend, don't reinvent.
@@ -0,0 +1,67 @@
1
+ Metadata-Version: 2.5
2
+ Name: gemini-router
3
+ Version: 0.1.1
4
+ Summary: Quota-aware Gemini Flash router: cascading fallback, per-key/model RPD/RPM/TPM ledger (SQLite WAL), key pool, wait policy, CLI
5
+ Author: vernikr
6
+ License: MIT
7
+ Keywords: gemini,llm,mcp,quota,rate-limit,router
8
+ Requires-Python: >=3.12
9
+ Requires-Dist: httpx>=0.27
10
+ Requires-Dist: pydantic>=2.7
11
+ Requires-Dist: pyyaml>=6.0
12
+ Requires-Dist: rich>=13.7
13
+ Provides-Extra: dev
14
+ Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
15
+ Requires-Dist: pytest>=8.0; extra == 'dev'
16
+ Requires-Dist: respx>=0.21; extra == 'dev'
17
+ Requires-Dist: ruff>=0.6; extra == 'dev'
18
+ Description-Content-Type: text/markdown
19
+
20
+ # gemini-router
21
+
22
+ Quota-aware **Gemini Flash router** as a standalone, reusable module — extracted and
23
+ generalized from [`tldr-digest/src/llm.py`](https://github.com/vernikr/tldr-digest).
24
+
25
+ One persistent ledger (SQLite WAL) = the single source of quota truth shared by every
26
+ consumer on a machine: the [gemini-mcp](https://github.com/vernikr/gemini-mcp) gateway,
27
+ `tldr-digest` cron runs, CLI calls, foreign scripts.
28
+
29
+ - Cascading chain with runtime `listModels` validation (default
30
+ `gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`; config-overridable).
31
+ - Per **key×model** budgets: RPD (Pacific-time day + rolling 24 h), RPM, TPM (sliding
32
+ 60 s) — pre-flight enforced, accounted from `usageMetadata`; conservative 429 handling
33
+ with cooldowns and `Retry-After`.
34
+ - Key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`), budget-maximizing selection.
35
+ - `thinking_level` per request (Gemini 3.x rules), request cache (0-RPD hits), opt-in
36
+ overflow tiers (`flash-lite → gemma-auto`), event stream for progress frontends.
37
+ - Per-request wait policy for minute-scale gates (RPM/TPM/short cooldowns): default
38
+ **wait** with countdown events (`waiting`, remaining seconds); `wait=False` skips to
39
+ the next candidate or returns wait ETAs in the structured error. Daily resets are
40
+ never waited on.
41
+ - CLI: `gemini-router quota | models | calls | prune | doctor` (rich tables).
42
+ - Deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3` — **no MCP/web-framework coupling**.
43
+
44
+ ## Status
45
+
46
+ **v0.1.1 implemented** (milestone M1 of the
47
+ [gemini-mcp plan](https://github.com/vernikr/gemini-mcp/blob/main/docs/04-plan.md)):
48
+ library + CLI, 45 tests green, CI on push. Contract: [`docs/SPEC.md`](docs/SPEC.md).
49
+
50
+ ## Planned usage
51
+
52
+ ```python
53
+ from gemini_router import Router, RouterRequest
54
+ router = Router.from_env()
55
+ res = await router.complete(RouterRequest(prompt=..., thinking_level="high"),
56
+ on_event=print)
57
+ ```
58
+
59
+ ```bash
60
+ uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
61
+ gemini-router quota # RPD left / RPM & TPM windows / cooldowns / next recommendation
62
+ ```
63
+
64
+ Service-style (no Python): REST `/v1/complete` + `/v1/quota` on the gemini-mcp daemon.
65
+
66
+ Conventions: English in files, Conventional Commits, SemVer tags `v*` (gemini-mcp pins
67
+ a tag), CI = ruff + pytest with mocked HTTP (see `AGENTS.md`).
@@ -0,0 +1,48 @@
1
+ # gemini-router
2
+
3
+ Quota-aware **Gemini Flash router** as a standalone, reusable module — extracted and
4
+ generalized from [`tldr-digest/src/llm.py`](https://github.com/vernikr/tldr-digest).
5
+
6
+ One persistent ledger (SQLite WAL) = the single source of quota truth shared by every
7
+ consumer on a machine: the [gemini-mcp](https://github.com/vernikr/gemini-mcp) gateway,
8
+ `tldr-digest` cron runs, CLI calls, foreign scripts.
9
+
10
+ - Cascading chain with runtime `listModels` validation (default
11
+ `gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`; config-overridable).
12
+ - Per **key×model** budgets: RPD (Pacific-time day + rolling 24 h), RPM, TPM (sliding
13
+ 60 s) — pre-flight enforced, accounted from `usageMetadata`; conservative 429 handling
14
+ with cooldowns and `Retry-After`.
15
+ - Key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`), budget-maximizing selection.
16
+ - `thinking_level` per request (Gemini 3.x rules), request cache (0-RPD hits), opt-in
17
+ overflow tiers (`flash-lite → gemma-auto`), event stream for progress frontends.
18
+ - Per-request wait policy for minute-scale gates (RPM/TPM/short cooldowns): default
19
+ **wait** with countdown events (`waiting`, remaining seconds); `wait=False` skips to
20
+ the next candidate or returns wait ETAs in the structured error. Daily resets are
21
+ never waited on.
22
+ - CLI: `gemini-router quota | models | calls | prune | doctor` (rich tables).
23
+ - Deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3` — **no MCP/web-framework coupling**.
24
+
25
+ ## Status
26
+
27
+ **v0.1.1 implemented** (milestone M1 of the
28
+ [gemini-mcp plan](https://github.com/vernikr/gemini-mcp/blob/main/docs/04-plan.md)):
29
+ library + CLI, 45 tests green, CI on push. Contract: [`docs/SPEC.md`](docs/SPEC.md).
30
+
31
+ ## Planned usage
32
+
33
+ ```python
34
+ from gemini_router import Router, RouterRequest
35
+ router = Router.from_env()
36
+ res = await router.complete(RouterRequest(prompt=..., thinking_level="high"),
37
+ on_event=print)
38
+ ```
39
+
40
+ ```bash
41
+ uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
42
+ gemini-router quota # RPD left / RPM & TPM windows / cooldowns / next recommendation
43
+ ```
44
+
45
+ Service-style (no Python): REST `/v1/complete` + `/v1/quota` on the gemini-mcp daemon.
46
+
47
+ Conventions: English in files, Conventional Commits, SemVer tags `v*` (gemini-mcp pins
48
+ a tag), CI = ruff + pytest with mocked HTTP (see `AGENTS.md`).
@@ -0,0 +1,126 @@
1
+ # gemini-router · SPEC (stage 2/3 artifact slice)
2
+
3
+ Date: 2026-10-08 · Companion docs live in `vernikr/gemini-mcp/docs/01..04` (system-wide
4
+ requirements, architecture, plan). This file is the router's own contract; updated as-built
5
+ each milestone.
6
+
7
+ ## 1. What it is
8
+
9
+ Quota-aware Gemini Flash router extracted from `tldr-digest/src/llm.py`, generalized:
10
+
11
+ - cascading model chain with runtime `listModels` validation (default
12
+ `gemini-3.8-flash → 3.7 → 3.6 → 3.5 → 3-flash-preview-slot`);
13
+ - per **key×model** budgets: RPD (America/Los_Angeles day + rolling 24 h), RPM and TPM
14
+ (sliding 60 s) — enforced pre-flight, accounted post-fact from `usageMetadata`;
15
+ - key pool (`GEMINI_API_KEYS` / `GEMINI_API_KEY_1..10`); selection prefers the pair with
16
+ the largest remaining budget;
17
+ - persistent multi-process ledger (SQLite WAL) — the single source of quota truth shared
18
+ by every consumer on the machine (MCP daemon, stdio sessions, cron jobs, foreign scripts);
19
+ - 429/5xx/4xx taxonomy inherited from tldr-digest (`RETRYABLE_STATUS`,
20
+ `MODEL_SWITCH_STATUS`, `Retry-After`, backoff+jitter, provider deadline);
21
+ - opt-in overflow tiers (`gemini-3.5-flash-lite`, best `gemma-*`) — never automatic;
22
+ - request cache (hash of model+params+prompt, TTL) — hits spend 0 RPD;
23
+ - event stream (`on_event`) for progress reporting by frontends;
24
+ - `thinking_level` (low|medium|high) per request; 3.x rules applied (no
25
+ temperature/top_p for Gemini 3.x; `thinking_level` not `thinking_budget`);
26
+ - CLI: `quota`, `models`, `calls`, `prune`, `doctor`; sync facade for legacy callers.
27
+
28
+ Zero MCP/web-framework deps: `httpx`, `pydantic`, `rich`, stdlib `sqlite3`. Python ≥3.12.
29
+
30
+ ## 2. Public API (v1 contract)
31
+
32
+ ```python
33
+ from gemini_router import Router, RouterRequest, RouterResult, RouterEvent, QuotaExhaustedError
34
+
35
+ router = Router.from_env() # env + optional GEMINI_ROUTER_CONFIG yaml
36
+ res: RouterResult = await router.complete(
37
+ RouterRequest(
38
+ prompt=str, system=str | None = None,
39
+ thinking_level="low|medium|high" = "high",
40
+ output_format="text|json" = "text",
41
+ max_output_tokens=8192,
42
+ kind="single|chunk|consolidate|count" = "single",
43
+ request_id=None, source="cli", overflow="off|lite|lite+gemma",
44
+ wait=True, # wait out minute-scale gates (RPM/TPM/short cooldown,
45
+ # ≤ wait.max_s), countdown via `waiting` events;
46
+ # daily RPD resets are NEVER waited on
47
+ model=None), # pin → skips chain fallback
48
+ on_event=callable | None) # RouterEvent(stage, pct, model, key_id, wait_s, detail, ts)
49
+ # stages: routing|waiting|dispatch|retry|switch|overflow|ok|quota_exhausted
50
+
51
+ res.text, res.model, res.key_id, res.tokens_in, res.tokens_out
52
+ res.latency_s, res.cached, res.overflow_tier, res.events, res.quota_after
53
+ await router.count_tokens(text, model=None) -> int
54
+ await router.quota_table() -> list[QuotaRow] # key_id, model, rpd_left, rpm_wait_s,
55
+ # tpm_wait_s, cooldown_until, recommended
56
+ router.complete_sync(...) # blocking facade (tldr-digest migration)
57
+ ```
58
+
59
+ Errors: `QuotaExhaustedError` carries the machine-readable payload
60
+ (per-model rpd_left/rpm_wait_s/tpm_wait_s/cooldown_until, reset_eta_pt, hint) — frontends
61
+ surface it as-is. `RouterError` base for everything else (network, http 4xx, parse).
62
+
63
+ ## 3. Ledger (SQLite, `GEMINI_ROUTER_STATE`, default `/opt/apps/shared/gemini-router/state.db`)
64
+
65
+ Tables `calls`, `cooldowns`, `cache`, `meta` — full DDL and accounting rules:
66
+ gemini-mcp `docs/03-architecture.md` §3 (single canonical definition).
67
+ Rules summary: WAL + busy_timeout 5 s; TPM pre-flight writes a `status='planned'` row,
68
+ upgraded after the call; RPD counts `ok|quota|http5xx` attempts (conservative),
69
+ not client-validation 4xx/network; PT-midnight + rolling-24h dual gate; retention prune
70
+ via `gemini-router prune` and by frontends' daily task.
71
+
72
+ ## 4. Config
73
+
74
+ Env: `GEMINI_API_KEYS`/`GEMINI_API_KEY_1..10`, `GEMINI_ROUTER_STATE`,
75
+ `GEMINI_ROUTER_CONFIG` (yaml). YAML keys: `chain`, `overflow_tiers`,
76
+ `limits_per_key_model {rpd,rpm,tpm}`, `overflow_limits {rpd,rpm,tpm}`, `limits_overrides`,
77
+ `quota_day_tz`, `retry {max,backoff_base_s,jitter_max_s,provider_timeout_s,request_delay_s}`,
78
+ `wait {enabled_default,max_s,poll_s}` (FR-C13 semantics), `window_s`, `planned_ttl_s`,
79
+ `models_refresh_h`, `cache_ttl_h`, `state_path`, `base_url`, `temperature` (pre-3.x models
80
+ only), `max_output_tokens_default`, `chars_per_token`. Defaults per gemini-mcp `docs/03-architecture.md` §8. Secrets never logged
81
+ (`key_id = sha256(key)[:8]`).
82
+
83
+ ## 5. Integration guide (external consumers)
84
+
85
+ ```bash
86
+ uv add "gemini-router @ git+https://github.com/vernikr/gemini-router.git@v1.0.0"
87
+ # private repo: export UV_INDEX / use gh auth or a deploy key; see README
88
+ ```
89
+
90
+ Service-style consumption (no Python): REST `POST /v1/complete`, `GET /v1/quota` on the
91
+ gemini-mcp daemon (localhost:8790, bearer) — same ledger, same rules.
92
+
93
+ ## 6. Extraction mapping from tldr-digest/src/llm.py
94
+
95
+ | tldr-digest | router | Change |
96
+ |---|---|---|
97
+ | `DEFAULT_MODEL_PRIORITY`/`llm_model_priority` | `config.chain` | Gemini-only in v1; groq/openrouter stay in tldr until a provider abstraction is justified |
98
+ | `_resolve_models`/`_list_models`/`_best_available` | `routing.resolve_chain` | + refresh interval, + preview-slot heuristics |
99
+ | `RETRYABLE_STATUS`/`MODEL_SWITCH_STATUS`/`_parse_retry_after`/backoff | `routing` | ported as-is |
100
+ | `Provider.can_call/register_call` (in-memory) | `ledger` queries | persistent, per key×model, + TPM |
101
+ | `_call_gemini` | `gemini.generate` | async httpx, thinking_level, usageMetadata accounting |
102
+ | `extract_json` | `gemini.extract_json` | ported as-is |
103
+ | `complete_json` | `complete(output_format="json")` | superset |
104
+
105
+ ## 7. Versioning & CI
106
+
107
+ SemVer; tags `v*`; gemini-mcp pins a tag. CI: ruff + pytest (respx mocks, no network) on
108
+ push/PR (free runners). Conventional Commits.
109
+
110
+ ## 8. As-built notes
111
+
112
+ - v0.1.1 (2026-10-08, live VPS smoke): **thinking-token starvation** discovered —
113
+ in 3.x models `maxOutputTokens` counts thinking tokens, so a tiny cap (16) yields
114
+ empty visible text and no `candidatesTokenCount`. Fixes: thought parts
115
+ (`thought=true`) never join visible text; `GenResult`/`RouterResult` carry
116
+ `finish_reason` + `tokens_thought`; `max_output_tokens_default` 8192→32768
117
+ (cap only, no spend implication). 45 tests green.
118
+ - v0.1.0 (M1): full contract above implemented; 43 tests green (ledger dual-gate RPD,
119
+ planned-row TPM windows, wait policy incl. cap and countdown events, cache +
120
+ request_id idempotency, overflow lite→gemma, live 429-day cooldown, multi-key pool,
121
+ countTokens spends no RPD, pinned-model handling).
122
+ - `extract_json` improvement over the tldr original: when several JSON candidates match,
123
+ the earliest starting in the text wins (fixes nested `{…[…]…}` extraction).
124
+ - Ledger statuses: `planned|ok|quota|http4xx|http5xx|network|blocked|skipped`;
125
+ RPD-conservative set = `ok,quota,http5xx,blocked` (D3).
126
+ - CLI exit codes: 0 ok, 1 doctor failed, 2 config/usage error.
@@ -0,0 +1,45 @@
1
+ [project]
2
+ name = "gemini-router"
3
+ version = "0.1.1"
4
+ description = "Quota-aware Gemini Flash router: cascading fallback, per-key/model RPD/RPM/TPM ledger (SQLite WAL), key pool, wait policy, CLI"
5
+ readme = "README.md"
6
+ requires-python = ">=3.12"
7
+ license = { text = "MIT" }
8
+ authors = [{ name = "vernikr" }]
9
+ keywords = ["gemini", "router", "quota", "rate-limit", "llm", "mcp"]
10
+ dependencies = [
11
+ "httpx>=0.27",
12
+ "pydantic>=2.7",
13
+ "pyyaml>=6.0",
14
+ "rich>=13.7",
15
+ ]
16
+
17
+ [project.optional-dependencies]
18
+ dev = [
19
+ "pytest>=8.0",
20
+ "pytest-asyncio>=0.24",
21
+ "respx>=0.21",
22
+ "ruff>=0.6",
23
+ ]
24
+
25
+ [project.scripts]
26
+ gemini-router = "gemini_router.cli:main"
27
+
28
+ [build-system]
29
+ requires = ["hatchling"]
30
+ build-backend = "hatchling.build"
31
+
32
+ [tool.hatch.build.targets.wheel]
33
+ packages = ["src/gemini_router"]
34
+
35
+ [tool.pytest.ini_options]
36
+ asyncio_mode = "auto"
37
+ testpaths = ["tests"]
38
+
39
+ [tool.ruff]
40
+ line-length = 100
41
+ target-version = "py312"
42
+ src = ["src", "tests"]
43
+
44
+ [tool.ruff.lint]
45
+ select = ["E", "F", "I", "UP", "B"]
@@ -0,0 +1,23 @@
1
+ """gemini-router — quota-aware Gemini Flash router with a persistent ledger.
2
+
3
+ Public API (docs/SPEC.md §2):
4
+ Router, RouterRequest, RouterResult, RouterEvent,
5
+ RouterError, GeminiHTTPError, QuotaExhaustedError,
6
+ Config, Limits, Ledger, GeminiClient, extract_json, key_id, estimate_tokens
7
+ """
8
+ from .config import Config, Limits, RetryCfg, WaitCfg, estimate_tokens, key_id
9
+ from .errors import GeminiHTTPError, QuotaExhaustedError, RouterError
10
+ from .events import RouterEvent
11
+ from .gemini import GeminiClient, GenResult, extract_json, is_gemini3
12
+ from .ledger import Ledger, iso, pt_midnight_ts
13
+ from .routing import Router, RouterRequest, RouterResult, cache_hash, gates_for, pick_gemma
14
+
15
+ __version__ = "0.1.1"
16
+
17
+ __all__ = [
18
+ "Router", "RouterRequest", "RouterResult", "RouterEvent",
19
+ "RouterError", "GeminiHTTPError", "QuotaExhaustedError",
20
+ "Config", "Limits", "RetryCfg", "WaitCfg", "Ledger", "GeminiClient", "GenResult",
21
+ "extract_json", "is_gemini3", "key_id", "estimate_tokens", "cache_hash",
22
+ "gates_for", "pick_gemma", "pt_midnight_ts", "iso", "__version__",
23
+ ]
@@ -0,0 +1,188 @@
1
+ """Terminal UX (FR-C12): quota | models | calls | prune | doctor. Works over SSH
2
+ against the shared state file without touching the daemon (FR-E7)."""
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import asyncio
7
+ import json
8
+
9
+ from rich.console import Console
10
+ from rich.table import Table
11
+
12
+ from . import __version__
13
+ from .config import Config
14
+ from .errors import RouterError
15
+ from .ledger import Ledger, iso, pt_midnight_ts
16
+ from .routing import Router, pick_gemma
17
+
18
+ console = Console()
19
+
20
+
21
+ def _load_cfg(args) -> Config:
22
+ over = {"state_path": args.state} if getattr(args, "state", None) else {}
23
+ return Config.load(getattr(args, "config", None), **over)
24
+
25
+
26
+ def _router(cfg: Config) -> Router:
27
+ return Router(cfg, ledger=Ledger(cfg.state_path))
28
+
29
+
30
+ def cmd_quota(cfg: Config, args) -> int:
31
+ rows = _router(cfg).quota_table()
32
+ if args.json:
33
+ print(json.dumps(rows, indent=1, default=str))
34
+ return 0
35
+ t = Table(title="Gemini quota (per key×model)", title_style="bold cyan", expand=True)
36
+ for col in ("key", "model", "RPD left", "RPM win", "TPM win", "cooldown", "next"):
37
+ t.add_column(col)
38
+ for r in rows:
39
+ wait = f"{r['wait_s']:.0f}s ({r['reason']})" if r["wait_s"] > 0 else "—"
40
+ style = "green" if r["rpd_left"] > 0 and r["wait_s"] <= 0 else (
41
+ "yellow" if r["rpd_left"] > 0 else "red")
42
+ rpm_w = wait if r["reason"] == "rpm" else ""
43
+ tpm_w = wait if r["reason"] == "tpm" else ""
44
+ t.add_row(r["key_id"], f"[{style}]{r['model']}[/{style}]",
45
+ f"{r['rpd_left']}/{r['rpd_limit']}",
46
+ f"{r['rpm_used']}/{r['rpm_limit']} {rpm_w}".strip(),
47
+ f"{r['tpm_used']}/{r['tpm_limit']} {tpm_w}".strip(),
48
+ iso(r["cooldown_until"])[:19] if r["cooldown_until"] else "—",
49
+ "★" if r["recommended"] else "")
50
+ console.print(t)
51
+ return 0
52
+
53
+
54
+ def cmd_models(cfg: Config, args) -> int:
55
+ async def _run() -> dict:
56
+ r = _router(cfg)
57
+ try:
58
+ return await r.availability() if not args.offline else {}
59
+ finally:
60
+ await r.aclose()
61
+ avail = asyncio.run(_run())
62
+ union: set[str] = set()
63
+ for s in avail.values():
64
+ if s:
65
+ union |= s
66
+ out = {"chain": cfg.chain, "overflow_tiers": cfg.overflow_tiers,
67
+ "gemma_pick": pick_gemma(union) if union else None,
68
+ "per_key": {k: (sorted(v) if v else "listModels unavailable → trust config")
69
+ for k, v in avail.items()},
70
+ "missing_from_live": sorted(m for m in cfg.chain if union and m not in union)}
71
+ if args.json:
72
+ print(json.dumps(out, indent=1))
73
+ return 0
74
+ console.print_json(json.dumps(out, indent=1))
75
+ return 0
76
+
77
+
78
+ def cmd_calls(cfg: Config, args) -> int:
79
+ rows = Ledger(cfg.state_path).calls_tail(
80
+ args.n, model=args.model, status=args.status, source=args.source,
81
+ request_id=args.request_id)
82
+ if args.json:
83
+ print(json.dumps(rows, indent=1, default=str))
84
+ return 0
85
+ t = Table(title=f"Last {len(rows)} calls", expand=True)
86
+ for col in ("ts (UTC)", "src", "kind", "key", "model", "status", "http",
87
+ "in", "out", "ms", "tier"):
88
+ t.add_column(col, overflow="fold")
89
+ for r in reversed(rows):
90
+ style = {"ok": "green", "quota": "red", "http5xx": "yellow",
91
+ "network": "yellow", "blocked": "magenta"}.get(r["status"], "dim")
92
+ t.add_row(r["ts_iso"][:19], r["source"], r["kind"], r["key_id"], r["model"],
93
+ f"[{style}]{r['status']}[/{style}]", str(r["http_status"] or ""),
94
+ str(r["tokens_in"] or ""), str(r["tokens_out"] or ""),
95
+ str(r["latency_ms"] or ""), str(r["overflow_tier"]))
96
+ console.print(t)
97
+ return 0
98
+
99
+
100
+ def cmd_prune(cfg: Config, args) -> int:
101
+ res = Ledger(cfg.state_path).prune(
102
+ audit_keep_days=args.audit_days, cache_ttl_s=cfg.cache_ttl_h * 3600, dry=args.dry)
103
+ console.print(("DRY " if args.dry else "") + json.dumps(res))
104
+ return 0
105
+
106
+
107
+ def cmd_doctor(cfg: Config, args) -> int:
108
+ ok = True
109
+ console.print(f"[bold]gemini-router {__version__} doctor[/bold]")
110
+ console.print(f" state: {cfg.state_path}")
111
+ try:
112
+ led = Ledger(cfg.state_path)
113
+ console.print(f" state stats: {led.stats()}")
114
+ except Exception as e: # noqa: BLE001
115
+ console.print(f" [red]state unusable: {e}[/red]")
116
+ return 1
117
+ if cfg.api_keys:
118
+ from .config import key_id
119
+ console.print(f" keys: {len(cfg.api_keys)} → "
120
+ + ", ".join(key_id(k) for k in cfg.api_keys))
121
+ else:
122
+ console.print(" [red]no API keys configured "
123
+ "(GEMINI_API_KEYS / GEMINI_API_KEY_1..10)[/red]")
124
+ ok = False
125
+ tz_now = pt_midnight_ts(cfg.quota_day_tz)
126
+ console.print(f" quota day ({cfg.quota_day_tz}): resets in "
127
+ f"{(pt_midnight_ts(cfg.quota_day_tz, days_ahead=1) - tz_now) / 3600:.1f} h "
128
+ f"(midnight was {iso(tz_now)[:19]} UTC)")
129
+ console.print(f" chain: {' → '.join(cfg.chain)}")
130
+ console.print(f" limits/key×model: {cfg.limits_per_key_model.model_dump()} · "
131
+ f"overflow: {cfg.overflow_limits.model_dump()} · "
132
+ f"wait: {cfg.wait.model_dump()}")
133
+ if not args.offline and cfg.api_keys:
134
+ async def _probe() -> None:
135
+ r = _router(cfg)
136
+ try:
137
+ avail = await r.availability()
138
+ for kid, models in avail.items():
139
+ if models is None:
140
+ console.print(f" [yellow]key {kid}: listModels failed (offline?)[/yellow]")
141
+ else:
142
+ have = [m for m in cfg.chain if m in models]
143
+ console.print(f" key {kid}: {len(models)} models live; chain available: "
144
+ + (", ".join(have) or "[red]NONE[/red]"))
145
+ finally:
146
+ await r.aclose()
147
+ asyncio.run(_probe())
148
+ console.print("[green]OK[/green]" if ok else "[red]FAILED[/red]")
149
+ return 0 if ok else 1
150
+
151
+
152
+ def main(argv: list[str] | None = None) -> int:
153
+ ap = argparse.ArgumentParser(prog="gemini-router",
154
+ description="Quota-aware Gemini Flash router CLI")
155
+ ap.add_argument("--config", help="YAML config path (or GEMINI_ROUTER_CONFIG)")
156
+ ap.add_argument("--state", help="override state db path (or GEMINI_ROUTER_STATE)")
157
+ ap.add_argument("--version", action="version", version=f"gemini-router {__version__}")
158
+ sub = ap.add_subparsers(dest="cmd", required=True)
159
+
160
+ p = sub.add_parser("quota", help="live budget table (which model can take a request)")
161
+ p.add_argument("--json", action="store_true")
162
+ p = sub.add_parser("models", help="resolved chain vs live listModels")
163
+ p.add_argument("--offline", action="store_true")
164
+ p.add_argument("--json", action="store_true")
165
+ p = sub.add_parser("calls", help="audit log tail")
166
+ p.add_argument("-n", type=int, default=50)
167
+ for flag in ("--model", "--status", "--source", "--request-id"):
168
+ p.add_argument(flag)
169
+ p.add_argument("--json", action="store_true")
170
+ p = sub.add_parser("prune", help="apply retention (audit days, cache TTL)")
171
+ p.add_argument("--audit-days", type=float, default=30)
172
+ p.add_argument("--dry", action="store_true")
173
+ p = sub.add_parser("doctor", help="config/state/connectivity check")
174
+ p.add_argument("--offline", action="store_true")
175
+
176
+ args = ap.parse_args(argv)
177
+ handlers = {"quota": cmd_quota, "models": cmd_models, "calls": cmd_calls,
178
+ "prune": cmd_prune, "doctor": cmd_doctor}
179
+ try:
180
+ cfg = _load_cfg(args)
181
+ return handlers[args.cmd](cfg, args)
182
+ except RouterError as e:
183
+ console.print(f"[red]error:[/red] {e}")
184
+ return 2
185
+
186
+
187
+ if __name__ == "__main__":
188
+ raise SystemExit(main())
@@ -0,0 +1,98 @@
1
+ """Config: pydantic model + YAML/env loading (SPEC §4). Defaults mirror
2
+ gemini-mcp docs/03 §8. Secrets come only from env; never from YAML."""
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import os
7
+ from pathlib import Path
8
+
9
+ import yaml
10
+ from pydantic import BaseModel, Field
11
+
12
+ DEFAULT_CHAIN = ["gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash",
13
+ "gemini-3.5-flash", "gemini-3-flash-preview"]
14
+ DEFAULT_OVERFLOW = ["gemini-3.5-flash-lite", "gemma-auto"]
15
+
16
+
17
+ class Limits(BaseModel):
18
+ rpd: int = 20
19
+ rpm: int = 5
20
+ tpm: int = 250_000
21
+
22
+
23
+ class RetryCfg(BaseModel):
24
+ max: int = 3
25
+ backoff_base_s: float = 5.0
26
+ jitter_max_s: float = 2.0
27
+ provider_timeout_s: float = 120.0
28
+ request_delay_s: float = 2.0
29
+
30
+
31
+ class WaitCfg(BaseModel):
32
+ enabled_default: bool = True # FR-C13: wait out minute-scale gates by default
33
+ max_s: float = 65.0
34
+ poll_s: float = 2.0
35
+
36
+
37
+ class Config(BaseModel):
38
+ chain: list[str] = Field(default_factory=lambda: list(DEFAULT_CHAIN))
39
+ overflow_tiers: list[str] = Field(default_factory=lambda: list(DEFAULT_OVERFLOW))
40
+ limits_per_key_model: Limits = Field(default_factory=Limits)
41
+ overflow_limits: Limits = Field(default_factory=lambda: Limits(rpd=500, rpm=15, tpm=250_000))
42
+ limits_overrides: dict[str, dict[str, int]] = Field(default_factory=dict)
43
+ quota_day_tz: str = "America/Los_Angeles"
44
+ retry: RetryCfg = Field(default_factory=RetryCfg)
45
+ wait: WaitCfg = Field(default_factory=WaitCfg)
46
+ window_s: float = 60.0 # sliding RPM/TPM window
47
+ planned_ttl_s: float = 300.0 # stale 'planned' rows stop counting after this
48
+ models_refresh_h: float = 12.0
49
+ cache_ttl_h: float = 24.0
50
+ state_path: str = "state.db"
51
+ base_url: str = "https://generativelanguage.googleapis.com/v1beta"
52
+ api_keys: list[str] = Field(default_factory=list)
53
+ temperature: float | None = None # applied to pre-3.x models only
54
+ max_output_tokens_default: int = 32768 # cap only; 3.x thinking tokens count toward it
55
+ chars_per_token: int = 4 # estimate when countTokens not provided
56
+
57
+ @classmethod
58
+ def load(cls, path: str | Path | None = None, **over) -> Config:
59
+ data: dict = {}
60
+ p = path or os.environ.get("GEMINI_ROUTER_CONFIG")
61
+ if p and Path(p).exists():
62
+ raw = yaml.safe_load(Path(p).read_text(encoding="utf-8")) or {}
63
+ data = raw.get("router", raw) if isinstance(raw.get("router"), dict) else raw
64
+ keys = _env_keys()
65
+ if keys:
66
+ data["api_keys"] = keys
67
+ if os.environ.get("GEMINI_ROUTER_STATE"):
68
+ data["state_path"] = os.environ["GEMINI_ROUTER_STATE"]
69
+ data.update({k: v for k, v in over.items() if v is not None})
70
+ return cls.model_validate(data)
71
+
72
+ def limits_for(self, model: str) -> Limits:
73
+ base = (self.limits_per_key_model if model in self.chain else self.overflow_limits)
74
+ merged = base.model_dump() | self.limits_overrides.get(model, {})
75
+ return Limits(**merged)
76
+
77
+
78
+ def _env_keys() -> list[str]:
79
+ keys: list[str] = []
80
+ v = os.environ.get("GEMINI_API_KEYS", "")
81
+ keys += [k.strip() for k in v.split(",") if k.strip()]
82
+ for i in range(1, 11):
83
+ k = os.environ.get(f"GEMINI_API_KEY_{i}", "").strip()
84
+ if k:
85
+ keys.append(k)
86
+ k = os.environ.get("GEMINI_API_KEY", "").strip()
87
+ if k:
88
+ keys.append(k)
89
+ return list(dict.fromkeys(keys))
90
+
91
+
92
+ def key_id(key: str) -> str:
93
+ """Masked key identity for logs/ledger (FR-E8): never store raw keys."""
94
+ return hashlib.sha256(key.encode()).hexdigest()[:8]
95
+
96
+
97
+ def estimate_tokens(text: str, chars_per_token: int = 4) -> int:
98
+ return max(1, len(text) // max(1, chars_per_token))