eggpool 0.3.6__tar.gz → 0.3.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {eggpool-0.3.6 → eggpool-0.3.7}/AGENTS.md +28 -12
- {eggpool-0.3.6 → eggpool-0.3.7}/CHANGELOG.md +23 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/PKG-INFO +33 -2
- {eggpool-0.3.6 → eggpool-0.3.7}/README.md +32 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/architecture/README.md +140 -5
- {eggpool-0.3.6 → eggpool-0.3.7}/config.example.toml +41 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/deployment.md +33 -5
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/network-diagnostics.md +29 -9
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/proxy.md +1 -1
- eggpool-0.3.7/docs/transcoding.md +329 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/pyproject.toml +4 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/check_database.py +1 -1
- eggpool-0.3.7/scripts/validate_routing.py +303 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/_share/config.example.toml +2 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/accounts/registry.py +49 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/accounts/state.py +22 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/backoff.py +20 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/network.py +17 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/proxy_request.py +151 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/stats.py +63 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/app.py +106 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/background/__init__.py +27 -39
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/cache.py +290 -38
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/catalog_resolvers.py +23 -5
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/service.py +2 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/cli_full.py +224 -11
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/constants.py +0 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/render.py +345 -66
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/routes.py +61 -6
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/static/dashboard.css +76 -17
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/static/dashboard.js +16 -11
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/theme.py +56 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/repositories.py +8 -3
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/rollup_repository.py +17 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0028_operational_events.sql +4 -6
- eggpool-0.3.7/src/eggpool/db/schema/0034_transcoding_daily.sql +17 -0
- eggpool-0.3.7/src/eggpool/db/schema/0035_routing_decision_score_components.sql +15 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/checksums.json +4 -2
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/deploy_user.py +1 -2
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/health/backoff.py +3 -5
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/health/health_manager.py +22 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/models/config.py +4 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/contract.py +17 -4
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/dns_cache.py +132 -17
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/outbound.py +25 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/proxy/usage.py +66 -35
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/quota/estimation.py +88 -23
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/quota/scorer.py +66 -8
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/body.py +3 -3
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/coordinator.py +914 -427
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/finalizer.py +20 -2
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/routing/eligibility.py +8 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/routing/router.py +278 -6
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/runtime.py +2 -4
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/runtime_dispatch.py +2 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/runtime_metrics.py +23 -47
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/stats/__init__.py +2 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/stats/grouped_timeseries.py +69 -34
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/stats/queries.py +241 -9
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/stats/service.py +50 -8
- eggpool-0.3.7/src/eggpool/transcoder/__init__.py +28 -0
- eggpool-0.3.7/src/eggpool/transcoder/anthropic_to_openai.py +259 -0
- eggpool-0.3.7/src/eggpool/transcoder/context.py +31 -0
- eggpool-0.3.7/src/eggpool/transcoder/errors.py +80 -0
- eggpool-0.3.7/src/eggpool/transcoder/ids.py +49 -0
- eggpool-0.3.7/src/eggpool/transcoder/openai_to_anthropic.py +284 -0
- eggpool-0.3.7/src/eggpool/transcoder/policy.py +40 -0
- eggpool-0.3.7/src/eggpool/transcoder/protocol.py +63 -0
- eggpool-0.3.7/src/eggpool/transcoder/static_headers.py +12 -0
- eggpool-0.3.7/src/eggpool/transcoder/streaming.py +743 -0
- eggpool-0.3.7/src/eggpool/transcoder/usage.py +82 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/uv.lock +1 -1
- {eggpool-0.3.6 → eggpool-0.3.7}/.env.example +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/.github/workflows/ci.yml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/.github/workflows/release.yml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/.gitignore +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/LICENSE +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/config-examples/claude-code.env +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/config-examples/opencode.jsonc +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/deploy/eggpool-logrotate.conf +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/deploy/eggpool.service +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/deploy/env.example +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/backup-restore.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/filesystem-layout.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/firewall.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/model-limits.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/providers.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/docs/raspberry-pi.md +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/install.sh +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/install_prompt.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/smoke_test.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/scripts/verify_upstream_auth.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/__main__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/_share/.env.example +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/accounts/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/chat_completions.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/errors.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/messages.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/models.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/runtime.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/api/update.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/auth.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/background/backup.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/background/cleanup.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/fetcher.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/limits.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/normalizer.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/pricing.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/pricing_aliases.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/pricing_resolver.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/catalog/protocols.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/cli.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/config.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/cost_recompute.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/_resources.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/escape.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/static/chart.umd.min.js +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/static/favicon.svg +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Booberry.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Catppuccin Latte.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Catppuccin Macchiato.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Catppuccin Mocha.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Cyber Red.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Cyberpunk.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Dark Green.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Discord (80_ Saturation).toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Discord.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Dracula.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Ferra Light.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Flexor Dark.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Gruvbox.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Halcyon Dark.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/IntelliJ Light.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Kanagawa.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Macaw Dark.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Macaw Light.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Matrix.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Noctis Lilac.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Nord.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Nostromo Terminal.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/One Dark.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Oxocarbon.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Rose Pine Dawn.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Rose Pine Moon.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Rose Pine.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Solarized Dark.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Sonokai.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Tokyo Night Storm.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/VESPER.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/Zenburn.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/acton.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/bam.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/base16-atelier-forest-light.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/berlin.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/black but with important highlights.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/broc.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/cork.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/ferra.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/forest.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/lisbon.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/midnight.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/oslo.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/plum.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/portland.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/sunset.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/tofino.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/vanimo.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/dashboard/themes/vik.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/connection.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/migrations.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0001_initial.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0002_indexes.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0003_request_attempts.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0004_integration_hardening.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0005_price_microdollars.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0006_correct_price_microdollars.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0007_price_cache_rates.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0008_proxy_request_identity.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0009_model_protocol_source.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0010_health_probe.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0011_model_resolution_status.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0012_drop_reservations_estimated_microdollars.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0013_request_attempts_account_id_index.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0014_bandwidth_tracking.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0015_multi_provider.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0016_requests_provider_id.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0017_price_snapshots_provider_id.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0018_provider_pings.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0019_client_ip.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0020_performance_indexes.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0021_provider_model_metadata.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0022_dashboard_indexes.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0023_deprecated_model_placeholder.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0024_account_backoffs.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0025_stale_request_index.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0026_attempt_observability.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0027_routing_decisions.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0029_latency_phases.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0030_model_pricing_aliases.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0031_price_snapshot_provenance.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0032_usage_rollups.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/db/schema/0033_request_provider_local_cost.sql +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/deploy/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/errors.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/fastcli.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/health/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/health/circuit_breaker.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/integrations/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/integrations/opencode.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/lifecycle/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/lifecycle/backup.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/lifecycle/uninstall.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/logging.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/metrics/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/metrics/buffer.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/models/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/models/api.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/models/database.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/models/domain.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/onboard.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/_templates.toml +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/auth.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/client_pool.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/connect.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/providers/pproxy_transport.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/proxy/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/proxy/client.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/proxy/cost_reporting.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/proxy/sse_observer.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/py.typed +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/quota/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/quota/audit.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/quota/reservation.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/attempt_finalizer.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/request/limits.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/retry/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/retry/classification.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/routing/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/routing/provider.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/runtime_paths.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/security/__init__.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/security/redaction.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/toml_edit.py +0 -0
- {eggpool-0.3.6 → eggpool-0.3.7}/src/eggpool/update_checker.py +0 -0
|
@@ -11,6 +11,7 @@ Project-specific skills are in `.opencode/skills/`:
|
|
|
11
11
|
## Quick Start
|
|
12
12
|
|
|
13
13
|
- Package manager: **uv** (not pip). Install deps: `uv sync --extra dev`
|
|
14
|
+
- CI installs with `uv sync --frozen --extra dev` (locks match `uv.lock` exactly)
|
|
14
15
|
- Entry point: `src/eggpool/cli.py` → `eggpool` console script
|
|
15
16
|
- Config: `config.toml` + `.env` for API keys
|
|
16
17
|
|
|
@@ -25,6 +26,17 @@ uv run pytest
|
|
|
25
26
|
|
|
26
27
|
All four must pass with zero errors.
|
|
27
28
|
|
|
29
|
+
## Focused Verification
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
uv run pytest tests/unit/test_contract.py -v # single test file
|
|
33
|
+
uv run pytest tests/unit/ -v # all unit tests
|
|
34
|
+
uv run pytest -k "test_something" -v # single test by name
|
|
35
|
+
uv run ruff check --fix src/ # auto-fix lint in one dir
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
CI sets `PYTHONHASHSEED=0` and `TZ=UTC`; reproduce locally for deterministic results.
|
|
39
|
+
|
|
28
40
|
## Code Style
|
|
29
41
|
|
|
30
42
|
- Python 3.11+ with `from __future__ import annotations` in ALL files
|
|
@@ -57,6 +69,7 @@ All four must pass with zero errors.
|
|
|
57
69
|
- **Request lifecycle**: `RequestCoordinator` orchestrates endpoint → routing → persistence → dispatch → finalization. See `architecture/README.md` § Request Lifecycle.
|
|
58
70
|
- **Multi-provider architecture**: provider-suffixed model IDs (`model-id/provider-id`), `ProviderClientPool`, `OutboundClientManager`. See `architecture/README.md` § Multi-Provider Architecture.
|
|
59
71
|
- **Provider contracts**: `compose_provider_url()` is the single source of truth for upstream URLs. See `architecture/README.md` § Provider Contracts and § Provider Contract Rendering.
|
|
72
|
+
- **Protocol transcoding**: transparent request/response format conversion between OpenAI and Anthropic protocols. Phase 2 body translation, Phase 3 streaming SSE translation, Phase 4 routing eligibility widening, and Phase 5 operator controls and docs are implemented in `src/eggpool/transcoder/` and `src/eggpool/request/coordinator.py`. See `architecture/README.md` § Protocol Transcoding.
|
|
60
73
|
- **Database invariants**: SQLite WAL, single-connection serialization, `async with db.transaction():` for all DML. See `architecture/README.md` § Database.
|
|
61
74
|
- **Quota and routing**: tier-based routing via `routing_priority`, `QuotaFairScorer`, upstream-authoritative suppression. See `architecture/README.md` § Quota and Routing.
|
|
62
75
|
- **Error hierarchy**: `AggregatorError` → `UpstreamError` → specific subclasses. See `architecture/README.md` § Error Hierarchy.
|
|
@@ -68,16 +81,21 @@ All four must pass with zero errors.
|
|
|
68
81
|
|
|
69
82
|
- Configuration changes require a service restart; live reload is intentionally not supported
|
|
70
83
|
- No pre-commit hooks are configured in this repo; CI runs ruff, pyright, and pytest via GitHub Actions
|
|
71
|
-
- **`static_models` is the source of truth for provider-specific protocol** — `FAMILY_PROTOCOLS` (`src/eggpool/catalog/protocols.py`) is a global fallback
|
|
72
|
-
- **Upstream-authoritative suppression**: local quota estimates are advisory by default (`local_quota_mode = "score_only"`).
|
|
73
|
-
- **Backoff persistence**: upstream-derived backoffs survive restarts via the `account_backoffs` table (`src/eggpool/db/schema/0024_account_backoffs.sql`).
|
|
74
|
-
- **Synthetic 503 vs 502**: `ModelUnavailableError` (503) is reserved for genuine pre-dispatch unavailability. `UpstreamExhaustedError` (502) is raised when every candidate account was attempted and exhausted mid-request.
|
|
75
|
-
- **Streaming finalizer shielding**: streaming `_build_stream_generator` finalization runs under `asyncio.shield(asyncio.wait_for(..., timeout=10))` so ASGI task cancellation cannot kill the finalizer while it holds the DB lock. Leaks that escape this path are caught by the periodic `stale_request_finalizer` background task (`app._finalize_stale_requests`, runs every 60s)
|
|
76
|
-
- **
|
|
77
|
-
-
|
|
78
|
-
- **`eggpool
|
|
79
|
-
- **
|
|
80
|
-
- **
|
|
84
|
+
- **`static_models` is the source of truth for provider-specific protocol** — `FAMILY_PROTOCOLS` (`src/eggpool/catalog/protocols.py`) is a global fallback. Providers like `minimax-cn` that serve MiniMax models on the OpenAI-compatible surface **must** ship `[[providers.<id>.static_models]]` rows with `protocol = "openai"`, otherwise the live `/v1/models` fetch resolves `MiniMax-M*` via the `minimax-` family prefix to `anthropic` and the protocol check clears it to `None`, producing `ModelUnavailableError` instead of `ProtocolMismatchError`. Static seeds survive via `ModelCatalogCache._preserve_static_fields` (`src/eggpool/catalog/cache.py:146-187`).
|
|
85
|
+
- **Upstream-authoritative suppression**: local quota estimates are advisory by default (`local_quota_mode = "score_only"`). Only upstream-observed failures (429/402/5xx/auth) and explicit operator disablement suppress routing. Switch to `hard_cap` only as an opt-in escape hatch.
|
|
86
|
+
- **Backoff persistence**: upstream-derived backoffs survive restarts via the `account_backoffs` table (`src/eggpool/db/schema/0024_account_backoffs.sql`). Local cost overruns must never be persisted as backoff rows.
|
|
87
|
+
- **Synthetic 503 vs 502**: `ModelUnavailableError` (503) is reserved for genuine pre-dispatch unavailability. `UpstreamExhaustedError` (502) is raised when every candidate account was attempted and exhausted mid-request.
|
|
88
|
+
- **Streaming finalizer shielding**: streaming `_build_stream_generator` finalization runs under `asyncio.shield(asyncio.wait_for(..., timeout=10))` so ASGI task cancellation cannot kill the finalizer while it holds the DB lock. Leaks that escape this path are caught by the periodic `stale_request_finalizer` background task (`app._finalize_stale_requests`, runs every 60s).
|
|
89
|
+
- **Routing-decision score components** (`RoutingDecisionTrace.score_components`) carry the per-account score breakdown on every persisted `routing_decisions` row (`score_components_json`, migration `0035`). The dashboard and `eggpool accounts explain` consume this directly; do not re-score from quota tables when only the diagnostic breakdown is needed.
|
|
90
|
+
- **`_select_lock` publish ordering**: in `RequestCoordinator._select_and_persist_attempt()`, runtime publication (`Router.increment_active_request_count` + `QuotaEstimator.add_reservation`) must run INSIDE `_select_lock` AFTER the durable transaction commits but BEFORE the lock releases. The two contexts are explicit nested `async with` blocks (outer `_select_lock`, inner `_db.transaction()`). Key invariant: block placement (publication outside DB transaction body, inside `_select_lock`), not context-exit order.
|
|
91
|
+
- **`eggpool accounts explain` reads from SQLite, not an empty cache**: the command hydrates the catalog via `ModelCatalogCache.hydrate_from_db(db)` from `models`/`provider_model_metadata`/`account_models` rows. A thin `_CatalogShim` exposes the loaded cache as a `CatalogService`-compatible object. Output uses `click.echo` (no `rich` dependency).
|
|
92
|
+
- **Startup crash recovery**: `_crash_recovery` runs at every startup and recovers ALL pending requests and active reservations with no time threshold. A process restart is a definitive boundary.
|
|
93
|
+
- **Pricing pipeline**: prices flow TOML override → upstream metadata → external catalog (OpenRouter / OpenCode Zen via the alias registry). Cost precedence: `provider_reported > derived/partial/exact > estimated > unknown`.
|
|
94
|
+
- **`eggpool stats recompute-costs [--dry-run|--apply] [--limit N]`**: recomputes cost from current price snapshots. Default `--dry-run`. Implemented in `src/eggpool/cost_recompute.py`.
|
|
95
|
+
- **Automatic backups**: in-process daily backups run by default under the `automatic_backup` supervised task (`src/eggpool/background/backup.py`). Controlled by `[backup]` config section.
|
|
96
|
+
- **DNS cache**: `OutboundClientManager` and `ProviderClientPool` both integrate a `DnsNetworkBackend` that caches resolved DNS entries. Controlled by `[network.dns_cache]` (enabled by default, TTL 1800s, max 50 entries). Exposes precise counters and derived rates for operator diagnostics.
|
|
97
|
+
- **Memory footprint caps**: every growth axis is bounded by hardcoded caps (`EWMA_HARD_CAP = 4096`, `GLOBAL_EWMA_HARD_CAP = 1024`, `MAX_TRACKED_HOSTS = 256`). Regression gate: `tests/integration/test_memory.py` (`pytest.mark.slow`). See `plans/memory.md` for the full design.
|
|
98
|
+
- **Transcoder body translation**: `select_transcoder()` in `src/eggpool/transcoder/protocol.py` is the single source of truth for translator dispatch. Loss-of-information warnings are accumulated on `TranscodeContext.loss_warnings` and logged at request completion.
|
|
81
99
|
|
|
82
100
|
## Error Handling
|
|
83
101
|
|
|
@@ -94,8 +112,6 @@ Use the hierarchy in `errors.py`. Chain exceptions with `raise ... from err` or
|
|
|
94
112
|
- **Do not add transitive imports to `runtime_paths` or `fastcli`** — they are stdlib-only and must stay lightweight for the Raspberry Pi watchdog contract
|
|
95
113
|
- Unrecognized commands fall through to `eggpool.cli_full`, which holds the heavy Click CLI
|
|
96
114
|
- Public symbols (`cli`, helpers used by tests) are lazily forwarded from `cli_full` via PEP 562 `__getattr__` — so `from eggpool.cli import cli` and existing test imports still work without loading the full graph
|
|
97
|
-
- See `plans/lightweight-cli-watchdog.md` for the full design
|
|
98
|
-
|
|
99
115
|
## Git Workflow
|
|
100
116
|
|
|
101
117
|
- Branch: `main`
|
|
@@ -7,6 +7,29 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- **Bidirectional OpenAI ↔ Anthropic protocol transcoding.** When `[transcoder] enabled = true`, requests from clients using one protocol can be forwarded to upstream accounts that speak only the other. Initial scope is text-only requests and responses, plus streaming SSE. Tool calls, vision, and extended thinking land in a follow-up release. See `docs/transcoding.md` for the full translation table.
|
|
13
|
+
- New `eggpool stats transcoding [--period 1d|7d|30d]` subcommand for transcoding observability.
|
|
14
|
+
- New "Transcoding" card on the `/runtime` dashboard page.
|
|
15
|
+
- Structured INFO log per transcoded request with `request_id`, protocol direction, account, and loss-warning count.
|
|
16
|
+
- Boot-time INFO line when `[transcoder] enabled = true` so operators see the configuration at startup.
|
|
17
|
+
- **`routing_decisions.score_components_json` column (migration `0035`)** carries the per-account score breakdown captured by `QuotaFairScorer` at the moment the coordinator chose the selected account. The dashboard can now answer "why account A over account B?" without rescoring from quota tables. Includes `quota_score`, `inflight_penalty`, `health_penalty`, `final_score`, `weight`, `active_request_count`, `reserved_microdollars`, per-window `cost_*` and `capacity_*` microdollar values, `tier`, `requires_transcode`, and the top 5 near-tie candidates.
|
|
18
|
+
- **`eggpool accounts explain --model <id> [--provider P] [--protocol P]`** subcommand (`src/eggpool/cli_full.py`) renders a Rich table listing every registered account with its live eligibility verdict and a stable `reason_code` (`disabled`, `auth_failed`, `quota_exhausted`, `cooldown`, `rate_limited`, `circuit_open`, `wrong_provider`, `no_protocol`, `protocol_mismatch`, `no_model`, `model_stale`, `ok`). Re-evaluated on every invocation against the live registry + catalog so operators can diagnose routing skew without restarting the service.
|
|
19
|
+
- **`GET /api/stats/routing/eligibility`** JSON counterpart (auth-gated via the existing stats-route dependency list, `src/eggpool/api/stats.py`) returns the same per-account verdict list as a JSON document for programmatic dashboards and alerting.
|
|
20
|
+
|
|
21
|
+
### Changed
|
|
22
|
+
|
|
23
|
+
- `RequestCoordinator` now carries `upstream_protocol` alongside `protocol` on `ProxyRequestContext`. Behaviour is identical when `[transcoder] enabled = false`.
|
|
24
|
+
- **`RequestCoordinator._select_and_persist_attempt()` lock scope.** The runtime publication step (`Router.increment_active_request_count` + `QuotaEstimator.add_reservation`) now runs INSIDE `_select_lock` AFTER the durable transaction commits but BEFORE the lock releases. The two contexts are written as explicit nested `async with` blocks (outer `_select_lock`, inner `_db.transaction()`). Note: collapsing them back into the previous compound `async with self._select_lock, self._db.transaction():` form would NOT by itself re-introduce the stale-score race on context-exit-order grounds — Python exits context managers right-to-left, so the transaction would still commit before the lock released. The actual bug was that the runtime publication block lived INSIDE the transaction body, so active-count and reserved-cost state were published before the durable transaction committed. The explicit nested form makes it hard to accidentally place publication inside the transaction while still keeping publication under `_select_lock`. The key invariant is block placement (publication must be outside the DB transaction body but still inside `_select_lock`), not context-exit order. The compensation chain (decrement → finalize-as-cancelled → release health slot → set `client_metadata["post_commit_interrupted"]` → re-raise) is preserved and still catches `BaseException` (including `CancelledError` / `SystemExit` / `KeyboardInterrupt`, all re-raised without being swallowed).
|
|
25
|
+
- `RoutingScore` gains diagnostic fields (`reserved_microdollars`, `cost_5h_microdollars`, `cost_7d_microdollars`, `cost_30d_microdollars`, `capacity_5h_microdollars`, `capacity_7d_microdollars`, `capacity_30d_microdollars`, `active_request_count`) so the scorer can return enough state to populate `score_components_json` without a second pass over the quota tables.
|
|
26
|
+
- `RoutingDecisionTrace` gains `score_components: Mapping[str, Any] | None` plus `to_score_components_json()`; `RoutingDecisionRepository.create()` accepts an optional `score_components_json` argument (defaults to `'{}'` for backward compatibility with rows inserted by code paths that have not yet been migrated).
|
|
27
|
+
- **`score_components_json` payload adds per-window utilization ratios and a tie-break summary.** The diagnostic JSON now carries `util_5h`, `util_7d`, `util_30d` (None when capacity is unconfigured) plus a `tie_break` dict naming the decisive factor between the chosen account and the runner-up (`tier`, `quota`, `inflight`, `transcode`, `near_tie`, `exact_tie`, `no_runner_up`) so the dashboard can surface a concrete cause without re-scoring.
|
|
28
|
+
- **`eggpool accounts explain` hydrates the catalog from SQLite.** The command now opens the database, runs migrations on a fresh install, and calls `ModelCatalogCache.hydrate_from_db(db)` (a new read-only helper on the cache module) to populate the in-memory model / provider / account-support tables from `models`, `provider_model_metadata`, and `account_models` rows before classification. The previous implementation constructed an empty cache and would have reported every account as ineligible even if the catalog-service shape had been right.
|
|
29
|
+
- **`eggpool accounts explain` no longer imports `rich`.** The undeclared `rich` dependency was replaced with plain `click.echo` columnar output. `reason_detail` strings now embed the account name, provider id, configured protocols, requested model id, and stale-window seconds so operators can act directly on the diagnosis.
|
|
30
|
+
- **`eggpool accounts status` now prints `routing_priority`.** The per-line output gained a `priority=N` field derived from the account's provider, alongside `provider`, `enabled`, `weight`, and the api-key-env set state.
|
|
31
|
+
- **`eggpool accounts explain` runs migrations on fresh installs.** The inner `_run_explain` coroutine now calls `MigrationRunner(db).run()` before hydrating `ModelCatalogCache.hydrate_from_db(db)`, so a brand-new (unmigrated) database path no longer crashes with `sqlite3.OperationalError: no such table: models` / `provider_model_metadata` / `account_models`. With no catalog rows yet, accounts surface a `no_model` verdict instead of the SQL error. The command still performs no outbound provider refresh.
|
|
32
|
+
|
|
10
33
|
## [0.3.5] - 2026-06-27
|
|
11
34
|
|
|
12
35
|
### Changed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: eggpool
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.7
|
|
4
4
|
Summary: A lightweight proxy that aggregates multiple LLM provider accounts behind one OpenAI-compatible endpoint
|
|
5
5
|
Project-URL: Homepage, https://github.com/eggstack/eggpool
|
|
6
6
|
Project-URL: Repository, https://github.com/eggstack/eggpool
|
|
@@ -60,6 +60,7 @@ A lightweight, LAN-hosted proxy that aggregates multiple AI provider accounts be
|
|
|
60
60
|
- Tracks requests, tokens, latency, errors, and estimated costs in SQLite
|
|
61
61
|
- Multi-page dashboard with 50+ themes, reliability, routing, and runtime views
|
|
62
62
|
- Designed for lightweight deployments (Raspberry Pi, SBCs)
|
|
63
|
+
- Transparent protocol transcoding between OpenAI and Anthropic request formats
|
|
63
64
|
|
|
64
65
|
## Quick Start
|
|
65
66
|
|
|
@@ -89,7 +90,9 @@ See [Deployment](docs/deployment.md) for alternative install methods (pipx, manu
|
|
|
89
90
|
| `eggpool rehash` | Restart to apply config changes |
|
|
90
91
|
| `eggpool stop` | Stop the running server |
|
|
91
92
|
| `eggpool models refresh` | Refresh the model catalog |
|
|
92
|
-
| `eggpool
|
|
93
|
+
| `eggpool stats transcoding` | Show protocol transcoding statistics |
|
|
94
|
+
| `eggpool accounts status` | Show configured account status (provider, priority, weight, enabled) |
|
|
95
|
+
| `eggpool accounts explain` | Show per-account routing eligibility for a model |
|
|
93
96
|
| `eggpool runtime-status` | Print runtime health summary |
|
|
94
97
|
| `eggpool backup` | Create a timestamped backup |
|
|
95
98
|
| `eggpool recover` | Restore from a backup archive |
|
|
@@ -131,9 +134,35 @@ Use `eggpool connect` for interactive provider setup. See [docs/providers.md](do
|
|
|
131
134
|
| `[dashboard]` | Dashboard toggle, theme, refresh interval |
|
|
132
135
|
| `[providers.*]` | Provider configs with accounts and routing priority |
|
|
133
136
|
| `[network]` | Outbound transport, DNS cache |
|
|
137
|
+
| `[transcoder]` | Protocol transcoding between OpenAI and Anthropic formats |
|
|
134
138
|
|
|
135
139
|
Full config reference: [`config.example.toml`](config.example.toml) | [docs/providers.md](docs/providers.md)
|
|
136
140
|
|
|
141
|
+
## Protocol transcoding
|
|
142
|
+
|
|
143
|
+
When `[transcoder] enabled = true`, EggPool bridges OpenAI Chat Completions and Anthropic Messages bidirectionally so a single client ecosystem (e.g. OpenCode, which speaks only OpenAI) can reach Anthropic-only upstreams (e.g. MiniMax International at `api.minimax.io/anthropic`) and vice versa.
|
|
144
|
+
|
|
145
|
+
What gets translated:
|
|
146
|
+
|
|
147
|
+
- Request bodies (text-only in v1)
|
|
148
|
+
- Streaming SSE events
|
|
149
|
+
- Non-retryable error envelopes
|
|
150
|
+
- Usage and cost fields (preserved exactly as the upstream reported them)
|
|
151
|
+
|
|
152
|
+
What is dropped with a structured warning log:
|
|
153
|
+
|
|
154
|
+
- OpenAI fields with no Anthropic equivalent (`logit_bias`, `presence_penalty`, `top_logprobs`, etc.)
|
|
155
|
+
- Anthropic fields with no OpenAI equivalent (`top_k`, `cache_control`)
|
|
156
|
+
|
|
157
|
+
What is **not** translated in v1 (lands in phase 6):
|
|
158
|
+
|
|
159
|
+
- Tool calls / function calling
|
|
160
|
+
- Vision / image content
|
|
161
|
+
- Extended thinking / reasoning
|
|
162
|
+
- Structured outputs (`response_format` / `json_schema`)
|
|
163
|
+
|
|
164
|
+
See [docs/transcoding.md](docs/transcoding.md) for the full translation table and known limitations.
|
|
165
|
+
|
|
137
166
|
## API Endpoints
|
|
138
167
|
|
|
139
168
|
| Method | Path | Description |
|
|
@@ -143,6 +172,7 @@ Full config reference: [`config.example.toml`](config.example.toml) | [docs/prov
|
|
|
143
172
|
| `POST` | `/v1/messages` | Anthropic-compatible messages |
|
|
144
173
|
| `GET` | `/v1/healthz` | Liveness check |
|
|
145
174
|
| `GET` | `/v1/readyz` | Readiness check |
|
|
175
|
+
| `GET` | `/api/backoffs` | Active upstream-derived account backoffs (`?now=<epoch>` for reproducible snapshots) |
|
|
146
176
|
|
|
147
177
|
When `[dashboard].enabled = true`, a multi-page dashboard is served at `/` with request stats, latency metrics, provider health, and more. Stats API available under `/api/stats/*`.
|
|
148
178
|
|
|
@@ -159,6 +189,7 @@ When `[dashboard].enabled = true`, a multi-page dashboard is served at `/` with
|
|
|
159
189
|
| Firewall configuration | [docs/firewall.md](docs/firewall.md) |
|
|
160
190
|
| Filesystem layout | [docs/filesystem-layout.md](docs/filesystem-layout.md) |
|
|
161
191
|
| Network & DNS diagnostics | [docs/network-diagnostics.md](docs/network-diagnostics.md) |
|
|
192
|
+
| Protocol transcoding | [docs/transcoding.md](docs/transcoding.md) |
|
|
162
193
|
|
|
163
194
|
## Development
|
|
164
195
|
|
|
@@ -16,6 +16,7 @@ A lightweight, LAN-hosted proxy that aggregates multiple AI provider accounts be
|
|
|
16
16
|
- Tracks requests, tokens, latency, errors, and estimated costs in SQLite
|
|
17
17
|
- Multi-page dashboard with 50+ themes, reliability, routing, and runtime views
|
|
18
18
|
- Designed for lightweight deployments (Raspberry Pi, SBCs)
|
|
19
|
+
- Transparent protocol transcoding between OpenAI and Anthropic request formats
|
|
19
20
|
|
|
20
21
|
## Quick Start
|
|
21
22
|
|
|
@@ -45,7 +46,9 @@ See [Deployment](docs/deployment.md) for alternative install methods (pipx, manu
|
|
|
45
46
|
| `eggpool rehash` | Restart to apply config changes |
|
|
46
47
|
| `eggpool stop` | Stop the running server |
|
|
47
48
|
| `eggpool models refresh` | Refresh the model catalog |
|
|
48
|
-
| `eggpool
|
|
49
|
+
| `eggpool stats transcoding` | Show protocol transcoding statistics |
|
|
50
|
+
| `eggpool accounts status` | Show configured account status (provider, priority, weight, enabled) |
|
|
51
|
+
| `eggpool accounts explain` | Show per-account routing eligibility for a model |
|
|
49
52
|
| `eggpool runtime-status` | Print runtime health summary |
|
|
50
53
|
| `eggpool backup` | Create a timestamped backup |
|
|
51
54
|
| `eggpool recover` | Restore from a backup archive |
|
|
@@ -87,9 +90,35 @@ Use `eggpool connect` for interactive provider setup. See [docs/providers.md](do
|
|
|
87
90
|
| `[dashboard]` | Dashboard toggle, theme, refresh interval |
|
|
88
91
|
| `[providers.*]` | Provider configs with accounts and routing priority |
|
|
89
92
|
| `[network]` | Outbound transport, DNS cache |
|
|
93
|
+
| `[transcoder]` | Protocol transcoding between OpenAI and Anthropic formats |
|
|
90
94
|
|
|
91
95
|
Full config reference: [`config.example.toml`](config.example.toml) | [docs/providers.md](docs/providers.md)
|
|
92
96
|
|
|
97
|
+
## Protocol transcoding
|
|
98
|
+
|
|
99
|
+
When `[transcoder] enabled = true`, EggPool bridges OpenAI Chat Completions and Anthropic Messages bidirectionally so a single client ecosystem (e.g. OpenCode, which speaks only OpenAI) can reach Anthropic-only upstreams (e.g. MiniMax International at `api.minimax.io/anthropic`) and vice versa.
|
|
100
|
+
|
|
101
|
+
What gets translated:
|
|
102
|
+
|
|
103
|
+
- Request bodies (text-only in v1)
|
|
104
|
+
- Streaming SSE events
|
|
105
|
+
- Non-retryable error envelopes
|
|
106
|
+
- Usage and cost fields (preserved exactly as the upstream reported them)
|
|
107
|
+
|
|
108
|
+
What is dropped with a structured warning log:
|
|
109
|
+
|
|
110
|
+
- OpenAI fields with no Anthropic equivalent (`logit_bias`, `presence_penalty`, `top_logprobs`, etc.)
|
|
111
|
+
- Anthropic fields with no OpenAI equivalent (`top_k`, `cache_control`)
|
|
112
|
+
|
|
113
|
+
What is **not** translated in v1 (lands in phase 6):
|
|
114
|
+
|
|
115
|
+
- Tool calls / function calling
|
|
116
|
+
- Vision / image content
|
|
117
|
+
- Extended thinking / reasoning
|
|
118
|
+
- Structured outputs (`response_format` / `json_schema`)
|
|
119
|
+
|
|
120
|
+
See [docs/transcoding.md](docs/transcoding.md) for the full translation table and known limitations.
|
|
121
|
+
|
|
93
122
|
## API Endpoints
|
|
94
123
|
|
|
95
124
|
| Method | Path | Description |
|
|
@@ -99,6 +128,7 @@ Full config reference: [`config.example.toml`](config.example.toml) | [docs/prov
|
|
|
99
128
|
| `POST` | `/v1/messages` | Anthropic-compatible messages |
|
|
100
129
|
| `GET` | `/v1/healthz` | Liveness check |
|
|
101
130
|
| `GET` | `/v1/readyz` | Readiness check |
|
|
131
|
+
| `GET` | `/api/backoffs` | Active upstream-derived account backoffs (`?now=<epoch>` for reproducible snapshots) |
|
|
102
132
|
|
|
103
133
|
When `[dashboard].enabled = true`, a multi-page dashboard is served at `/` with request stats, latency metrics, provider health, and more. Stats API available under `/api/stats/*`.
|
|
104
134
|
|
|
@@ -115,6 +145,7 @@ When `[dashboard].enabled = true`, a multi-page dashboard is served at `/` with
|
|
|
115
145
|
| Firewall configuration | [docs/firewall.md](docs/firewall.md) |
|
|
116
146
|
| Filesystem layout | [docs/filesystem-layout.md](docs/filesystem-layout.md) |
|
|
117
147
|
| Network & DNS diagnostics | [docs/network-diagnostics.md](docs/network-diagnostics.md) |
|
|
148
|
+
| Protocol transcoding | [docs/transcoding.md](docs/transcoding.md) |
|
|
118
149
|
|
|
119
150
|
## Development
|
|
120
151
|
|
|
@@ -17,6 +17,7 @@ src/eggpool/
|
|
|
17
17
|
├── models/ # Pydantic config, domain, API, and database models
|
|
18
18
|
├── providers/ # ProviderClientPool, pproxy transport, connect CLI
|
|
19
19
|
├── proxy/ # Transparent proxy, SSE observer, usage extraction
|
|
20
|
+
├── transcoder/ # Protocol transcoding (OpenAI ↔ Anthropic, body + streaming)
|
|
20
21
|
├── quota/ # Quota estimation, reservations, scoring
|
|
21
22
|
├── request/ # RequestCoordinator, finalizers, body reader, limit enforcement
|
|
22
23
|
├── retry/ # Error classification and failover
|
|
@@ -49,7 +50,8 @@ All data-plane requests flow through `RequestCoordinator`:
|
|
|
49
50
|
2. **Routing** selects an eligible account via quota-aware scoring (`routing/router.py`)
|
|
50
51
|
3. **Attempt** is persisted to SQLite before upstream dispatch
|
|
51
52
|
4. **Provider Contract** renders absolute URL (`compose_provider_url()`) and auth headers (`build_upstream_headers()`) from `providers/contract.py`
|
|
52
|
-
5. **
|
|
53
|
+
5. **Protocol Transcoding** (if enabled) translates the request body when the client protocol differs from the upstream protocol
|
|
54
|
+
6. **Proxy** sends the request via the provider's `httpx.AsyncClient` from `ProviderClientPool`
|
|
53
55
|
6. **Streaming** is handled by `proxy/sse_observer.py` with chunk-level usage extraction
|
|
54
56
|
7. **Finalization** records usage, releases reservations, updates health state
|
|
55
57
|
|
|
@@ -62,7 +64,7 @@ Key invariants:
|
|
|
62
64
|
- Each attempt reservation is released exactly once via `AttemptFinalizer`
|
|
63
65
|
- The same URL composition rules apply to catalog fetch and chat dispatch
|
|
64
66
|
- **Structured observability persistence (migrations 0026-0029)** every `request_attempts` row carries provider/model/protocol/retry_category/latency/bytes/streamed/is_retry_outcome; every routing decision is persisted to `routing_decisions` in the same transaction as the `request_attempts` INSERT; safety-net tasks (`_crash_recovery`, `_finalize_stale_requests_once`, `reconcile_expired_reservations`) record `operational_events` rows inside the same transaction as the durable state mutation; latency is decomposed into `upstream_connect_ms / upstream_read_ms / coordinator_overhead_ms` so the dashboard can distinguish network vs upstream vs eggpool-side bottlenecks
|
|
65
|
-
- **Runtime metrics are best-effort and process-local** — the `/api/stats/runtime` endpoint and `eggpool runtime-status` CLI command gather process topology, memory, background task state, database health, OS load average (`os.getloadavg` + normalized per-core), and a bounded rolling-window dispatch-overhead distribution via `DispatchOverheadRecorder` (`src/eggpool/runtime_dispatch.py`); failed probes return `null` rather than raising, and the endpoint is always auth-gated even with a public dashboard
|
|
67
|
+
- **Runtime metrics are best-effort and process-local** — the `/api/stats/runtime` endpoint and `eggpool runtime-status` CLI command gather process topology, memory, background task state, database health, OS load average (`os.getloadavg` + normalized per-core), and a bounded rolling-window dispatch-overhead distribution via `DispatchOverheadRecorder` (`src/eggpool/runtime_dispatch.py`); failed probes return `null` rather than raising, `probe_errors` is capped to 16 truncated entries, and the endpoint is always auth-gated even with a public dashboard
|
|
66
68
|
|
|
67
69
|
## Multi-Provider Architecture
|
|
68
70
|
|
|
@@ -124,6 +126,52 @@ The coordinator calls `_build_upstream_headers()` and `_get_upstream_url()` whic
|
|
|
124
126
|
|
|
125
127
|
`AppConfig.validate_account_credentials()` rejects API keys that begin with the `Bearer` scheme for providers configured with `auth.mode = "bearer"`. EggPool adds the scheme automatically, so a stored `Bearer <token>` would produce `Authorization: Bearer Bearer <token>` upstream and cause 401s. The same guard runs in `scripts/verify_upstream_auth.py` so the operator gets an explicit error before any upstream call. Providers using `auth.mode = "raw_authorization"` are unaffected because they pass the value verbatim.
|
|
126
128
|
|
|
129
|
+
## Protocol Transcoding
|
|
130
|
+
|
|
131
|
+
When a client sends a request in one protocol (e.g., Anthropic Messages API)
|
|
132
|
+
but the routed provider only supports another (e.g., OpenAI Chat Completions
|
|
133
|
+
API), the `transcoder` module translates the request body before dispatch and
|
|
134
|
+
the response body (including streaming chunks) after receipt.
|
|
135
|
+
|
|
136
|
+
**Phase 1 (foundation)** lands the data model, configuration surface, and
|
|
137
|
+
helper modules without changing runtime behaviour:
|
|
138
|
+
- `TranscoderPolicy` config model (`[transcoder]` section)
|
|
139
|
+
- `TranscodeContext` per-request state dataclass
|
|
140
|
+
- `upstream_protocol` field on `ProxyRequestContext`
|
|
141
|
+
- Mechanical refactor: upstream-side reads in the coordinator use
|
|
142
|
+
`context.upstream_protocol` instead of `context.protocol`
|
|
143
|
+
- Routing eligibility accepts a `transcode_eligibility` parameter
|
|
144
|
+
- Helper modules: `ids.py` (tool-call ID map), `usage.py` (usage
|
|
145
|
+
canonicalisation), `errors.py` (upstream error envelope parser)
|
|
146
|
+
|
|
147
|
+
**Phase 2 — Body translation**: text-only, non-streaming request/response
|
|
148
|
+
body translation is implemented in `src/eggpool/transcoder/`. The
|
|
149
|
+
`BodyTranscoder` Protocol (`protocol.py`) defines the interface;
|
|
150
|
+
`OpenAIToAnthropic` and `AnthropicToOpenAI` are the concrete translators.
|
|
151
|
+
`select_transcoder()` is the single source of truth for dispatch. The
|
|
152
|
+
coordinator pre-translates the request body before dispatch, decodes the
|
|
153
|
+
response body on success, and re-renders non-retryable errors in the client
|
|
154
|
+
protocol. Loss-of-information warnings are accumulated on
|
|
155
|
+
`TranscodeContext.loss_warnings` and logged at request completion.
|
|
156
|
+
|
|
157
|
+
**Phase 3 — Streaming translation**: SSE stream translation in both
|
|
158
|
+
directions for text-only streams. `StreamingTranscoder` implementations
|
|
159
|
+
(`OpenAIToAnthropicStreaming`, `AnthropicToOpenAIStreaming`) translate
|
|
160
|
+
upstream SSE frames into client-format bytes chunk-by-chunk.
|
|
161
|
+
`select_streaming_transcoder()` in `streaming.py` is the dispatch source
|
|
162
|
+
of truth. The coordinator's `_build_stream_generator` applies the transcoder
|
|
163
|
+
when the client and upstream protocols differ. Same-protocol requests pass
|
|
164
|
+
through unchanged. Tool calls, thinking, and routing widening are out of
|
|
165
|
+
scope (phases 4–6).
|
|
166
|
+
|
|
167
|
+
**Phase 4 — Routing eligibility widening**: when `[transcoder] enabled = true`, the routing layer widens the candidate set to include accounts whose `provider.protocols` includes the model's native protocol even if it does not include the client protocol. `_validate_endpoint` checks for transcodable routes before raising `ProtocolMismatchError`. The `_resolve_upstream_protocol` method determines which protocol to use upstream based on the largest eligible-account set. `prefer_native = true` (default) keeps native-protocol accounts ranked above transcodable ones via a secondary sort key in `QuotaFairScorer`. The two-pass context-limit check in `api/proxy_request.py` validates both client-side and upstream limits when transcoding is active.
|
|
168
|
+
|
|
169
|
+
**Phase 5 — Operator controls and docs**: the default `[transcoder]` config block is documented and uncommented in `config.example.toml`. `eggpool stats transcoding` reports transcoded request counts and loss-warning summaries. The dashboard `/runtime` page includes a "Transcoding" card showing real-time counters. Structured INFO logs are emitted for every transcoded request and at boot time when enabled. See `docs/transcoding.md` for the full operator guide.
|
|
170
|
+
|
|
171
|
+
Token counts are mapped between protocol-specific fields (e.g.,
|
|
172
|
+
`input_tokens` → `prompt_tokens`, `cache_creation_input_tokens` →
|
|
173
|
+
separate cache counters). Controlled by `[transcoder]` config.
|
|
174
|
+
|
|
127
175
|
## Database
|
|
128
176
|
|
|
129
177
|
SQLite via aiosqlite with WAL mode. Single-connection serialization via a lock + ContextVar.
|
|
@@ -137,7 +185,7 @@ SQLite via aiosqlite with WAL mode. Single-connection serialization via a lock +
|
|
|
137
185
|
|
|
138
186
|
### Schema Migrations
|
|
139
187
|
|
|
140
|
-
Ordered SQL migrations in `db/schema/` (0001 through
|
|
188
|
+
Ordered SQL migrations in `db/schema/` (0001 through 0035). Checksums tracked in `checksums.json`.
|
|
141
189
|
|
|
142
190
|
### Repositories
|
|
143
191
|
|
|
@@ -183,8 +231,7 @@ Accounts are excluded from routing when:
|
|
|
183
231
|
In the default `score_only` mode, local cost and quota estimates influence
|
|
184
232
|
routing **priority** only — above-capacity accounts stay eligible. Only
|
|
185
233
|
upstream-observed failures, explicit operator disablement, and catalog/
|
|
186
|
-
protocol incompatibility can suppress routing.
|
|
187
|
-
`plans/upstream-authoritative-suppression.md` for the full design.
|
|
234
|
+
protocol incompatibility can suppress routing.
|
|
188
235
|
|
|
189
236
|
Upstream-derived backoffs (429, 402, model-unavailable) persist across
|
|
190
237
|
restarts in the `account_backoffs` table (`src/eggpool/db/schema/0024_account_backoffs.sql`)
|
|
@@ -198,6 +245,76 @@ the coordinator raises `UpstreamExhaustedError` (502) — synthetic 503 is
|
|
|
198
245
|
reserved for genuine pre-dispatch unavailability (no enabled accounts,
|
|
199
246
|
missing credentials, all explicitly disabled, model unknown).
|
|
200
247
|
|
|
248
|
+
### Lock scope and publish ordering
|
|
249
|
+
|
|
250
|
+
The `RequestCoordinator._select_and_persist_attempt()` method holds
|
|
251
|
+
`_select_lock` across both the durable transaction (`request_attempts`
|
|
252
|
+
+ `routing_decisions` INSERT inside `async with self._db.transaction():`)
|
|
253
|
+
AND the runtime publication step (`Router.increment_active_request_count`
|
|
254
|
+
+ `QuotaEstimator.add_reservation`). The publication runs AFTER the
|
|
255
|
+
transaction commits but BEFORE the lock releases, so a concurrent
|
|
256
|
+
selector that enters the lock next observes this attempt's runtime
|
|
257
|
+
state. The publish is fast (in-process counter + cache mutation), so
|
|
258
|
+
the lock-hold stays tight while still closing the burst-skew race
|
|
259
|
+
previously caused by publishing inside the transaction body.
|
|
260
|
+
|
|
261
|
+
Note: the two contexts are written as explicit nested `async with`
|
|
262
|
+
blocks (outer `_select_lock`, inner `_db.transaction()`) — NOT as a
|
|
263
|
+
compound `async with self._select_lock, self._db.transaction():`. A
|
|
264
|
+
compound form would still exit right-to-left (transaction commits
|
|
265
|
+
before the lock releases), so context-exit order alone is not the
|
|
266
|
+
invariant. The actual bug was that the runtime publication block lived
|
|
267
|
+
INSIDE the transaction body; active-count and reserved-cost state were
|
|
268
|
+
therefore published before the transaction committed. The explicit
|
|
269
|
+
nested form makes it hard to accidentally place publication inside the
|
|
270
|
+
transaction while still keeping publication under `_select_lock`. The
|
|
271
|
+
key invariant is block placement (publication must be outside the DB
|
|
272
|
+
transaction body but still inside `_select_lock`), not context-exit
|
|
273
|
+
order.
|
|
274
|
+
|
|
275
|
+
The compensation chain (`decrement` → finalize-as-cancelled → release
|
|
276
|
+
health slot → set `client_metadata["post_commit_interrupted"]` →
|
|
277
|
+
re-raise) wraps the publish step and catches `BaseException` (including
|
|
278
|
+
`CancelledError` / `SystemExit` / `KeyboardInterrupt`, re-raised
|
|
279
|
+
without swallowing).
|
|
280
|
+
|
|
281
|
+
### Score components and eligibility diagnostics
|
|
282
|
+
|
|
283
|
+
Every persisted `routing_decisions` row carries the per-account score
|
|
284
|
+
breakdown captured by `QuotaFairScorer` at the moment the coordinator
|
|
285
|
+
chose the selected account. Migration `0035` adds the
|
|
286
|
+
`score_components_json` column; `RoutingDecisionTrace.to_score_components_json()`
|
|
287
|
+
serializes the diagnostic payload (TEXT JSON, defaults to `'{}'` on rows
|
|
288
|
+
written by code paths that pre-date the migration). The payload now also
|
|
289
|
+
includes per-window `util_5h` / `util_7d` / `util_30d` utilization ratios
|
|
290
|
+
(None when the scorer's capacity is unconfigured) and a `tie_break`
|
|
291
|
+
summary naming the decisive factor between the chosen account and its
|
|
292
|
+
runner-up (`tier`, `quota`, `inflight`, `transcode`, `near_tie`,
|
|
293
|
+
`exact_tie`, `no_runner_up`). The same data flows through
|
|
294
|
+
`eggpool accounts explain --model <id> [--provider P] [--protocol P]`
|
|
295
|
+
and `GET /api/stats/routing/eligibility` for live operator diagnostics.
|
|
296
|
+
|
|
297
|
+
`Router.explain_account_eligibility(model_id, provider_id, protocol)`
|
|
298
|
+
returns one row per registered account with `eligible: bool`, a stable
|
|
299
|
+
`reason_code` (`ok`, `disabled`, `auth_failed`, `quota_exhausted`,
|
|
300
|
+
`cooldown`, `rate_limited`, `circuit_open`, `wrong_provider`,
|
|
301
|
+
`no_protocol`, `protocol_mismatch`, `no_model`, `model_stale`), and a
|
|
302
|
+
short `reason_detail` that names the account, its provider, its
|
|
303
|
+
configured protocols, the requested model id, and the stale-window
|
|
304
|
+
seconds (so the operator can act directly on the diagnosis). The
|
|
305
|
+
classification mirrors the live filter chain in
|
|
306
|
+
`eggpool.routing.eligibility.get_eligible_accounts` so explanations
|
|
307
|
+
match the routing path exactly.
|
|
308
|
+
|
|
309
|
+
`eggpool accounts explain` opens the database, runs migrations on a
|
|
310
|
+
fresh install, and calls `ModelCatalogCache.hydrate_from_db(db)` to
|
|
311
|
+
populate the in-memory model / provider / account-support tables from
|
|
312
|
+
the durable `models`, `provider_model_metadata`, and `account_models`
|
|
313
|
+
rows. The cache is wrapped in a tiny `_CatalogShim` so `Router` can
|
|
314
|
+
consume it without booting a full `CatalogService`; output is rendered
|
|
315
|
+
with `click.echo` (the previous `rich` table was removed because the
|
|
316
|
+
dependency was undeclared).
|
|
317
|
+
|
|
201
318
|
## Provider Routing Priority and Model Collapse
|
|
202
319
|
|
|
203
320
|
Two related configuration knobs let operators control how requests for the
|
|
@@ -469,3 +586,21 @@ Production (`eggpool deploy systemd --install --production`):
|
|
|
469
586
|
| `metrics_flush` | `write_mode != "immediate"` | Buffered analytics flush to `usage_rollups` |
|
|
470
587
|
| `update_checker` | Always | Periodic PyPI update check (default 24h); see `update_checker.py` |
|
|
471
588
|
| `automatic_backup` | `backup.enabled` | In-process SQLite backup with count-based retention; see `background/backup.py` |
|
|
589
|
+
| `health_disabled_models_prune` | Always | Periodic sweep that drops stale `model_availability` and `disabled_models` entries (every 60s) |
|
|
590
|
+
|
|
591
|
+
## In-Memory Bounds and Memory Footprint
|
|
592
|
+
|
|
593
|
+
Long-running deployments — especially Raspberry Pi / SBC nodes — must keep steady-state RSS bounded by workload throughput, not workload cardinality. Every growth axis in the hot path is capped by a hardcoded module constant or a per-catalog config knob; see `plans/memory.md` for the full design and the per-request regression test (`tests/integration/test_memory.py`, marked `pytest.mark.slow`).
|
|
594
|
+
|
|
595
|
+
| Structure | Location | Cap | Eviction |
|
|
596
|
+
|-----------|----------|-----|----------|
|
|
597
|
+
| `QuotaEstimator.account_model_ewma` | `src/eggpool/quota/estimation.py:285` | `EWMA_HARD_CAP = 4096` (hardcoded) | LRU; on miss recomputes from persisted `QuotaWindow` |
|
|
598
|
+
| `QuotaEstimator.global_model_ewma` | `src/eggpool/quota/estimation.py:286` | `GLOBAL_EWMA_HARD_CAP = 1024` (hardcoded) | LRU |
|
|
599
|
+
| `CatalogResolverPipeline.TTLCache._data` | `src/eggpool/catalog/catalog_resolvers.py:128` | `max_entries = 4096` per `[pricing.catalogs.<name>]` (configurable) | LRU on store; `entry.raw` stripped after parse |
|
|
600
|
+
| `ModelCatalogCache._models` / `_provider_models` | `src/eggpool/catalog/cache.py:109-111` | De-duplicated (per-provider override only when it differs from global) | — |
|
|
601
|
+
| `ModelCatalogCache._account_support` | `src/eggpool/catalog/cache.py:114` | `frozenset[str]` (no per-call `.copy()`); bounded by registered account × model cardinality | — |
|
|
602
|
+
| `OutboundClientManager._per_host_*` | `src/eggpool/providers/outbound.py:85` | `MAX_TRACKED_HOSTS = 256` (hardcoded) | Coldest-total eviction; `evictions_total` surfaced in `snapshot()` and the `outbound_client` runtime metric |
|
|
603
|
+
| `AccountRuntimeState.model_availability` | `src/eggpool/accounts/state.py` | Pruned at every `AccountRegistry.sync_accounts` against advertised model set | — |
|
|
604
|
+
| `HealthManager.AccountHealth.disabled_models` | `src/eggpool/health/health_manager.py:111` | Pruned by `health_disabled_models_prune` supervisor task (60s cycle) | — |
|
|
605
|
+
|
|
606
|
+
The `frozenset` switch on `_account_support` (`src/eggpool/catalog/cache.py:639`) eliminates one O(n) `set.copy()` per routing decision. Every caller of `get_supporting_accounts(...)` / `get_supporting_accounts_for_model(...)` is read-only (membership, intersection, iteration), so the immutability is a strict superset of caller needs.
|
|
@@ -39,6 +39,23 @@ ping_retain_days = 7
|
|
|
39
39
|
# # ID and routed across all of them. Default false
|
|
40
40
|
# # emits one provider-suffixed entry per provider.
|
|
41
41
|
|
|
42
|
+
# ─── Protocol Transcoding ────────────────────────────────────────────
|
|
43
|
+
# When enabled, requests from clients using one protocol (OpenAI or
|
|
44
|
+
# Anthropic) can be forwarded to upstream accounts whose provider
|
|
45
|
+
# declares only the other protocol. EggPool translates the request body,
|
|
46
|
+
# the streaming SSE events, and the response body bidirectionally.
|
|
47
|
+
#
|
|
48
|
+
# Use this when you want to mix protocol-native and protocol-transcoded
|
|
49
|
+
# accounts (e.g. opencode-go native plus MiniMax International
|
|
50
|
+
# Anthropic-only, both reachable from OpenCode clients).
|
|
51
|
+
#
|
|
52
|
+
# Default: enabled = false. The behaviour before this flag existed is
|
|
53
|
+
# preserved exactly.
|
|
54
|
+
[transcoder]
|
|
55
|
+
enabled = false
|
|
56
|
+
loss_policy = "warn" # "warn" or "reject" for loss-of-information
|
|
57
|
+
prefer_native = true # prefer native-protocol accounts in routing
|
|
58
|
+
|
|
42
59
|
[routing]
|
|
43
60
|
strategy = "quota_fair"
|
|
44
61
|
near_tie_epsilon = 0.1
|
|
@@ -63,6 +80,29 @@ five_hour_microdollars = 12000000
|
|
|
63
80
|
weekly_microdollars = 30000000
|
|
64
81
|
monthly_microdollars = 60000000
|
|
65
82
|
|
|
83
|
+
# --- Pricing resolution ----------------------------------------------
|
|
84
|
+
# External pricing catalogs supplement upstream model metadata when
|
|
85
|
+
# providers do not include pricing in /v1/models responses. The per-catalog
|
|
86
|
+
# max_entries cap bounds steady-state RSS on SBC deployments.
|
|
87
|
+
[pricing]
|
|
88
|
+
fallback = "generic_estimate"
|
|
89
|
+
|
|
90
|
+
[pricing.catalogs]
|
|
91
|
+
|
|
92
|
+
[pricing.catalogs.openrouter]
|
|
93
|
+
enabled = true
|
|
94
|
+
priority = 100
|
|
95
|
+
ttl_seconds = 86400
|
|
96
|
+
# max_entries = 4096 # bound the in-memory catalog cache; oldest entries evict first
|
|
97
|
+
# base_url = "https://openrouter.ai/api/v1"
|
|
98
|
+
# api_key = "..." # optional; only needed for higher rate limits
|
|
99
|
+
|
|
100
|
+
[pricing.catalogs.opencode_zen]
|
|
101
|
+
enabled = false
|
|
102
|
+
priority = 200
|
|
103
|
+
ttl_seconds = 86400
|
|
104
|
+
# max_entries = 4096 # bound the in-memory catalog cache; oldest entries evict first
|
|
105
|
+
|
|
66
106
|
# ─── Network ─────────────────────────────────────────────────────────
|
|
67
107
|
# Outbound client transport tuning and DNS caching. The DNS cache wraps
|
|
68
108
|
# the default httpcore transport to cache resolved DNS entries in memory,
|
|
@@ -77,7 +117,7 @@ monthly_microdollars = 60000000
|
|
|
77
117
|
# [network.dns_cache]
|
|
78
118
|
# enabled = true # enable in-memory DNS cache (default: true)
|
|
79
119
|
# max_entries = 50 # max cached host entries (default: 50)
|
|
80
|
-
# positive_ttl_seconds =
|
|
120
|
+
# positive_ttl_seconds = 1800 # upper cap for positive DNS answers (default: 1800)
|
|
81
121
|
# negative_ttl_seconds = 30 # how long to cache lookup failures (default: 30)
|
|
82
122
|
# stale_if_error_seconds = 3600 # max age for stale records usable on resolver failure (default: 3600)
|
|
83
123
|
# prefer_ipv6 = false # prefer IPv6 addresses in DNS resolution (default: false)
|
|
@@ -156,7 +156,6 @@ the operator's interactive shell environment. `ensure-running`
|
|
|
156
156
|
atomically checks-and-starts without ever spawning a duplicate
|
|
157
157
|
instance, and uses the stdlib-only fast-path CLI so the cron tick is
|
|
158
158
|
cheap enough to run every five minutes on Raspberry Pi-class hardware.
|
|
159
|
-
See `plans/lightweight-cli-watchdog.md` for the design rationale.
|
|
160
159
|
|
|
161
160
|
### 5. Manual run (debug foreground)
|
|
162
161
|
|
|
@@ -533,8 +532,19 @@ The output covers:
|
|
|
533
532
|
- **Routing** — active requests, pending count, active reservations,
|
|
534
533
|
reserved microdollars, health states, active backoff rows.
|
|
535
534
|
|
|
535
|
+
For a focused view of active upstream-derived suppression, call
|
|
536
|
+
`GET /api/backoffs`. The endpoint returns persisted `account_backoffs`
|
|
537
|
+
rows joined with account names and accepts `?now=<epoch seconds>` for
|
|
538
|
+
reproducible snapshots during tests or incident review. Local quota
|
|
539
|
+
estimates never appear in this response; only upstream-observed
|
|
540
|
+
failures such as 429, 402, auth failures, model unavailability, and
|
|
541
|
+
bounded transient errors create these rows.
|
|
542
|
+
|
|
536
543
|
All probes are best-effort; failed probes return `null` rather than
|
|
537
|
-
causing the command to fail.
|
|
544
|
+
causing the command to fail. Probe diagnostics are exposed in the JSON
|
|
545
|
+
payload as `probe_errors`, capped at 16 entries with each message
|
|
546
|
+
truncated, so repeated host or permissions failures cannot produce an
|
|
547
|
+
unbounded response.
|
|
538
548
|
|
|
539
549
|
### Checking from cron
|
|
540
550
|
|
|
@@ -635,7 +645,25 @@ section. On Linux the snapshot reads current RSS from
|
|
|
635
645
|
`/proc/self/stat`; on macOS it falls back to `ru_maxrss` (a
|
|
636
646
|
high-water mark).
|
|
637
647
|
|
|
638
|
-
|
|
648
|
+
The following in-memory growth axes are bounded by design; see
|
|
649
|
+
`plans/memory.md` for the full design:
|
|
650
|
+
|
|
651
|
+
- `QuotaEstimator.account_model_ewma` and `global_model_ewma` are
|
|
652
|
+
LRU-capped at `EWMA_HARD_CAP = 4096` and `GLOBAL_EWMA_HARD_CAP = 1024`
|
|
653
|
+
entries respectively (hardcoded, not configurable).
|
|
654
|
+
- `ModelCatalogCache` deduplicates `_models` and `_provider_models`,
|
|
655
|
+
and `_account_support` is a `frozenset[str]` (no per-call `.copy()`).
|
|
656
|
+
- `CatalogResolverPipeline.TTLCache` is bounded per catalog by
|
|
657
|
+
`max_entries` (default `4096`, configurable per `[pricing.catalogs.<name>]`).
|
|
658
|
+
- `OutboundClientManager._per_host_requests` / `_per_host_errors`
|
|
659
|
+
are capped at `MAX_TRACKED_HOSTS = 256` (coldest-total eviction;
|
|
660
|
+
`evictions_total` is exposed in the manager snapshot).
|
|
661
|
+
- `AccountRuntimeState.model_availability` and
|
|
662
|
+
`HealthManager.AccountHealth.disabled_models` are pruned at every
|
|
663
|
+
`AccountRegistry.sync_accounts` / `health_disabled_models_prune`
|
|
664
|
+
sweep against the currently-advertised model set.
|
|
665
|
+
|
|
666
|
+
If RSS still grows continuously after the above:
|
|
639
667
|
|
|
640
668
|
1. Check for leaked pending requests (see above).
|
|
641
669
|
2. Verify WAL checkpointing is working: `PRAGMA wal_checkpoint(PASSIVE)`.
|
|
@@ -680,7 +708,7 @@ A process restart is a definitive boundary: any request that was still
|
|
|
680
708
|
active reservations are released, regardless of how recently they were
|
|
681
709
|
created. Check the startup log for `Crash recovery: marked N stale
|
|
682
710
|
requests` to confirm a clean recovery after a crash or forced
|
|
683
|
-
restart.
|
|
711
|
+
restart.
|
|
684
712
|
|
|
685
713
|
---
|
|
686
714
|
|
|
@@ -708,4 +736,4 @@ restart. See `plans/eggpoolfix.md` for the full safety-net design.
|
|
|
708
736
|
| `eggpool uninstall` | Detect install method, preview PATH edits, remove binary + config + data + shell-rc entries |
|
|
709
737
|
| `eggpool uninstall --deploy-artifacts` | Also remove systemd unit, logrotate config, watchdog + backup cron blocks, backup script |
|
|
710
738
|
| `eggpool croncheck` | Check if server is running (exit 0/1) — fast-path, no heavy imports |
|
|
711
|
-
| `eggpool ensure-running` | Atomically check-and-start the server; no-op when already alive — fast-path, no heavy imports |
|
|
739
|
+
| `eggpool ensure-running` | Atomically check-and-start the server; no-op when already alive — fast-path, no heavy imports |
|