tollfree 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tollfree-0.1.0/.gitignore +11 -0
- tollfree-0.1.0/PKG-INFO +79 -0
- tollfree-0.1.0/README.md +60 -0
- tollfree-0.1.0/pyproject.toml +45 -0
- tollfree-0.1.0/src/tollfree/__init__.py +9 -0
- tollfree-0.1.0/src/tollfree/_registry.py +17 -0
- tollfree-0.1.0/src/tollfree/cli.py +23 -0
- tollfree-0.1.0/src/tollfree/probe.py +88 -0
- tollfree-0.1.0/src/tollfree/providers.json +60 -0
- tollfree-0.1.0/src/tollfree/router.py +115 -0
tollfree-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: tollfree
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: One chat() call over free-tier LLM providers, with automatic model + provider failover.
|
|
5
|
+
Project-URL: Homepage, https://github.com/luarss/free-llm-router
|
|
6
|
+
Project-URL: Source, https://github.com/luarss/free-llm-router
|
|
7
|
+
Author-email: luarss <song.luar@a5x.ai>
|
|
8
|
+
License: MIT
|
|
9
|
+
Keywords: failover,free-tier,groq,llm,openai,router
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Requires-Python: >=3.9
|
|
14
|
+
Requires-Dist: python-dotenv>=1.0
|
|
15
|
+
Requires-Dist: requests>=2.31
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest>=8.0; extra == 'test'
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# tollfree (Python)
|
|
21
|
+
|
|
22
|
+
One `chat()` call over free-tier LLM providers, with automatic failover:
|
|
23
|
+
**rotate a provider's models first, then fall over to the next provider.**
|
|
24
|
+
|
|
25
|
+
Keys are read straight from `.env`. Missing keys are skipped automatically — you
|
|
26
|
+
only need one to start.
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install tollfree # or: uv add tollfree
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Use
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from tollfree import chat
|
|
38
|
+
|
|
39
|
+
reply = chat("Explain the CAP theorem in two sentences.")
|
|
40
|
+
|
|
41
|
+
# with metadata about which provider/model actually answered:
|
|
42
|
+
reply, meta = chat("hi", return_meta=True, temperature=0.2)
|
|
43
|
+
# meta -> {'provider': 'google', 'model': 'gemini-3.6-flash'}
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
CLI:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
tollfree "your prompt here" # chat
|
|
50
|
+
tollfree probe # verify keys + limits, spends no completion tokens
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Failover logic
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
for provider in PROVIDERS: # priority order (providers.json)
|
|
57
|
+
for model in provider.models: # default = first
|
|
58
|
+
try -> return on success
|
|
59
|
+
401/403 -> skip whole provider (bad key)
|
|
60
|
+
429/404/5xx/timeout -> next model, then next provider
|
|
61
|
+
raise AllProvidersFailed # only when everything is exhausted
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Configuration
|
|
65
|
+
|
|
66
|
+
Set keys in `.env` (or the environment):
|
|
67
|
+
|
|
68
|
+
```
|
|
69
|
+
GROQ_API_KEY=
|
|
70
|
+
CEREBRAS_API_KEY=
|
|
71
|
+
GEMINI_API_KEY=
|
|
72
|
+
OPENROUTER_API_KEY=
|
|
73
|
+
MISTRAL_API_KEY=
|
|
74
|
+
ZAI_API_KEY=
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The provider/model registry is `providers.json`, kept in sync from the
|
|
78
|
+
[central repo](https://github.com/luarss/free-llm-router) which is the source of
|
|
79
|
+
truth for both the Python and npm packages.
|
tollfree-0.1.0/README.md
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# tollfree (Python)
|
|
2
|
+
|
|
3
|
+
One `chat()` call over free-tier LLM providers, with automatic failover:
|
|
4
|
+
**rotate a provider's models first, then fall over to the next provider.**
|
|
5
|
+
|
|
6
|
+
Keys are read straight from `.env`. Missing keys are skipped automatically — you
|
|
7
|
+
only need one to start.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install tollfree # or: uv add tollfree
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Use
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from tollfree import chat
|
|
19
|
+
|
|
20
|
+
reply = chat("Explain the CAP theorem in two sentences.")
|
|
21
|
+
|
|
22
|
+
# with metadata about which provider/model actually answered:
|
|
23
|
+
reply, meta = chat("hi", return_meta=True, temperature=0.2)
|
|
24
|
+
# meta -> {'provider': 'google', 'model': 'gemini-3.6-flash'}
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
CLI:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
tollfree "your prompt here" # chat
|
|
31
|
+
tollfree probe # verify keys + limits, spends no completion tokens
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Failover logic
|
|
35
|
+
|
|
36
|
+
```
|
|
37
|
+
for provider in PROVIDERS: # priority order (providers.json)
|
|
38
|
+
for model in provider.models: # default = first
|
|
39
|
+
try -> return on success
|
|
40
|
+
401/403 -> skip whole provider (bad key)
|
|
41
|
+
429/404/5xx/timeout -> next model, then next provider
|
|
42
|
+
raise AllProvidersFailed # only when everything is exhausted
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Configuration
|
|
46
|
+
|
|
47
|
+
Set keys in `.env` (or the environment):
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
GROQ_API_KEY=
|
|
51
|
+
CEREBRAS_API_KEY=
|
|
52
|
+
GEMINI_API_KEY=
|
|
53
|
+
OPENROUTER_API_KEY=
|
|
54
|
+
MISTRAL_API_KEY=
|
|
55
|
+
ZAI_API_KEY=
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
The provider/model registry is `providers.json`, kept in sync from the
|
|
59
|
+
[central repo](https://github.com/luarss/free-llm-router) which is the source of
|
|
60
|
+
truth for both the Python and npm packages.
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tollfree"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "One chat() call over free-tier LLM providers, with automatic model + provider failover."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "luarss", email = "song.luar@a5x.ai" }]
|
|
13
|
+
keywords = ["llm", "openai", "groq", "failover", "router", "free-tier"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
]
|
|
19
|
+
dependencies = [
|
|
20
|
+
"requests>=2.31",
|
|
21
|
+
"python-dotenv>=1.0",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[project.optional-dependencies]
|
|
25
|
+
test = ["pytest>=8.0"]
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
Homepage = "https://github.com/luarss/free-llm-router"
|
|
29
|
+
Source = "https://github.com/luarss/free-llm-router"
|
|
30
|
+
|
|
31
|
+
[project.scripts]
|
|
32
|
+
tollfree = "tollfree.cli:main"
|
|
33
|
+
|
|
34
|
+
[tool.hatch.build.targets.wheel]
|
|
35
|
+
packages = ["src/tollfree"]
|
|
36
|
+
|
|
37
|
+
[tool.hatch.build.targets.wheel.force-include]
|
|
38
|
+
"src/tollfree/providers.json" = "tollfree/providers.json"
|
|
39
|
+
|
|
40
|
+
[tool.hatch.build.targets.sdist]
|
|
41
|
+
include = ["src/tollfree", "README.md"]
|
|
42
|
+
|
|
43
|
+
[tool.pytest.ini_options]
|
|
44
|
+
testpaths = ["tests"]
|
|
45
|
+
addopts = "-q"
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""tollfree — one chat() call over free-tier LLM providers, with automatic
|
|
2
|
+
model + provider failover. See https://github.com/luarss/free-llm-router."""
|
|
3
|
+
|
|
4
|
+
from ._registry import PROVIDERS
|
|
5
|
+
from .probe import probe, probe_all
|
|
6
|
+
from .router import AllProvidersFailed, chat
|
|
7
|
+
|
|
8
|
+
__all__ = ["chat", "probe", "probe_all", "PROVIDERS", "AllProvidersFailed"]
|
|
9
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Load the canonical provider registry bundled with the package.
|
|
2
|
+
|
|
3
|
+
The registry lives in providers.json (kept in sync from the repo root by
|
|
4
|
+
scripts/sync-providers). Loading it here keeps a single source of truth.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
from importlib.resources import files
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def load_providers():
|
|
12
|
+
"""Return the provider list from the bundled providers.json."""
|
|
13
|
+
raw = files(__package__).joinpath("providers.json").read_text(encoding="utf-8")
|
|
14
|
+
return json.loads(raw)["providers"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
PROVIDERS = load_providers()
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Console entry point: `tollfree "prompt"` to chat, `tollfree probe` to check keys."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
from .probe import probe_all
|
|
6
|
+
from .router import chat
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main(argv=None):
|
|
10
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
11
|
+
|
|
12
|
+
if argv and argv[0] == "probe":
|
|
13
|
+
probe_all()
|
|
14
|
+
return
|
|
15
|
+
|
|
16
|
+
q = " ".join(argv) or "Say hello in one short sentence."
|
|
17
|
+
answer, meta = chat(q, return_meta=True, verbose=True)
|
|
18
|
+
print(f"\n--- {meta['provider']} / {meta['model']} ---")
|
|
19
|
+
print(answer)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
if __name__ == "__main__":
|
|
23
|
+
main()
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Probe providers WITHOUT spending completion quota.
|
|
3
|
+
|
|
4
|
+
For each provider that has a key in .env, this:
|
|
5
|
+
- GET /models -> verifies the key + auth, lists available models
|
|
6
|
+
(free; does not consume chat/token quota)
|
|
7
|
+
- prints any rate-limit headers the provider returns (real remaining quota)
|
|
8
|
+
- for OpenRouter, also GET /key for credit/limit/usage
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
|
|
13
|
+
import requests
|
|
14
|
+
from dotenv import load_dotenv
|
|
15
|
+
|
|
16
|
+
from ._registry import PROVIDERS
|
|
17
|
+
|
|
18
|
+
load_dotenv()
|
|
19
|
+
|
|
20
|
+
# Header names providers use to advertise limits/quota. We match loosely.
|
|
21
|
+
_LIMIT_HINTS = ("ratelimit", "rate-limit", "retry-after", "x-request", "quota")
|
|
22
|
+
|
|
23
|
+
_TIMEOUT = 20
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _limit_headers(headers):
|
|
27
|
+
out = {}
|
|
28
|
+
for k, v in headers.items():
|
|
29
|
+
lk = k.lower()
|
|
30
|
+
if any(h in lk for h in _LIMIT_HINTS):
|
|
31
|
+
out[k] = v
|
|
32
|
+
return out
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def probe(provider):
|
|
36
|
+
"""Probe a single provider dict; prints a status line (and any limit headers)."""
|
|
37
|
+
key = os.getenv(provider["key_env"])
|
|
38
|
+
name = provider["name"]
|
|
39
|
+
if not key:
|
|
40
|
+
print(f"— {name:12s} SKIP (no {provider['key_env']} in .env)")
|
|
41
|
+
return
|
|
42
|
+
|
|
43
|
+
try:
|
|
44
|
+
r = requests.get(
|
|
45
|
+
f"{provider['base_url']}/models",
|
|
46
|
+
headers={"Authorization": f"Bearer {key}"},
|
|
47
|
+
timeout=_TIMEOUT,
|
|
48
|
+
)
|
|
49
|
+
except requests.RequestException as e:
|
|
50
|
+
print(f"✘ {name:12s} NET ({e})")
|
|
51
|
+
return
|
|
52
|
+
|
|
53
|
+
if r.status_code == 200:
|
|
54
|
+
try:
|
|
55
|
+
models = r.json().get("data", [])
|
|
56
|
+
n = len(models)
|
|
57
|
+
except ValueError:
|
|
58
|
+
n = "?"
|
|
59
|
+
print(f"✔ {name:12s} OK key valid, {n} models visible")
|
|
60
|
+
elif r.status_code in (401, 403):
|
|
61
|
+
print(f"✘ {name:12s} AUTH HTTP {r.status_code} — bad/expired key")
|
|
62
|
+
else:
|
|
63
|
+
print(f"⚠ {name:12s} HTTP {r.status_code} — {r.text[:80]}")
|
|
64
|
+
|
|
65
|
+
for hk, hv in _limit_headers(r.headers).items():
|
|
66
|
+
print(f" · {hk}: {hv}")
|
|
67
|
+
|
|
68
|
+
# OpenRouter exposes exact limit/usage/credit for free.
|
|
69
|
+
if name == "openrouter" and r.status_code == 200:
|
|
70
|
+
try:
|
|
71
|
+
kr = requests.get(
|
|
72
|
+
"https://openrouter.ai/api/v1/key",
|
|
73
|
+
headers={"Authorization": f"Bearer {key}"},
|
|
74
|
+
timeout=_TIMEOUT,
|
|
75
|
+
).json().get("data", {})
|
|
76
|
+
print(
|
|
77
|
+
f" · usage=${kr.get('usage')} limit=${kr.get('limit')} "
|
|
78
|
+
f"free_tier={kr.get('is_free_tier')}"
|
|
79
|
+
)
|
|
80
|
+
except requests.RequestException:
|
|
81
|
+
pass
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def probe_all():
|
|
85
|
+
"""Probe every provider in the registry."""
|
|
86
|
+
print("Probing providers (no completion tokens spent)\n")
|
|
87
|
+
for p in PROVIDERS:
|
|
88
|
+
probe(p)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Canonical provider registry — single source of truth for both the PyPI and npm packages. Priority order matters (top = tried first). Every provider speaks the OpenAI-compatible /chat/completions API, so the only per-provider data is base_url, the env var holding the key, and the model list (first model = default). To add a provider: add a block and set its key env var. Run scripts/sync-providers to copy this into both packages.",
|
|
3
|
+
"providers": [
|
|
4
|
+
{
|
|
5
|
+
"name": "groq",
|
|
6
|
+
"base_url": "https://api.groq.com/openai/v1",
|
|
7
|
+
"key_env": "GROQ_API_KEY",
|
|
8
|
+
"models": [
|
|
9
|
+
"llama-3.3-70b-versatile",
|
|
10
|
+
"llama-3.1-8b-instant",
|
|
11
|
+
"qwen/qwen3-32b"
|
|
12
|
+
]
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"name": "cerebras",
|
|
16
|
+
"base_url": "https://api.cerebras.ai/v1",
|
|
17
|
+
"key_env": "CEREBRAS_API_KEY",
|
|
18
|
+
"models": [
|
|
19
|
+
"llama-3.3-70b",
|
|
20
|
+
"llama3.1-8b",
|
|
21
|
+
"qwen-3-32b"
|
|
22
|
+
]
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"name": "google",
|
|
26
|
+
"base_url": "https://generativelanguage.googleapis.com/v1beta/openai",
|
|
27
|
+
"key_env": "GEMINI_API_KEY",
|
|
28
|
+
"models": [
|
|
29
|
+
"gemini-3.6-flash",
|
|
30
|
+
"gemini-3.5-flash-lite"
|
|
31
|
+
]
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"name": "openrouter",
|
|
35
|
+
"base_url": "https://openrouter.ai/api/v1",
|
|
36
|
+
"key_env": "OPENROUTER_API_KEY",
|
|
37
|
+
"models": [
|
|
38
|
+
"meta-llama/llama-3.3-70b-instruct:free",
|
|
39
|
+
"google/gemini-2.0-flash-exp:free"
|
|
40
|
+
]
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"name": "mistral",
|
|
44
|
+
"base_url": "https://api.mistral.ai/v1",
|
|
45
|
+
"key_env": "MISTRAL_API_KEY",
|
|
46
|
+
"models": [
|
|
47
|
+
"mistral-small-latest",
|
|
48
|
+
"open-mistral-nemo"
|
|
49
|
+
]
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
"name": "zai",
|
|
53
|
+
"base_url": "https://api.z.ai/api/paas/v4",
|
|
54
|
+
"key_env": "ZAI_API_KEY",
|
|
55
|
+
"models": [
|
|
56
|
+
"glm-4-flash"
|
|
57
|
+
]
|
|
58
|
+
}
|
|
59
|
+
]
|
|
60
|
+
}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""
|
|
2
|
+
tollfree router: one chat() call, automatic model + provider failover.
|
|
3
|
+
|
|
4
|
+
Rotation order:
|
|
5
|
+
1. Within a provider, rotate through its models (default first).
|
|
6
|
+
2. When a provider's models are all exhausted, fall over to the next provider.
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
from tollfree import chat
|
|
10
|
+
reply = chat("Explain CAP theorem in two sentences.")
|
|
11
|
+
print(reply)
|
|
12
|
+
|
|
13
|
+
# or with full message history / options
|
|
14
|
+
reply, meta = chat(
|
|
15
|
+
messages=[{"role": "user", "content": "hi"}],
|
|
16
|
+
temperature=0.2,
|
|
17
|
+
return_meta=True,
|
|
18
|
+
)
|
|
19
|
+
print(meta) # {'provider': 'groq', 'model': 'llama-3.3-70b-versatile'}
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import os
|
|
23
|
+
|
|
24
|
+
import requests
|
|
25
|
+
from dotenv import load_dotenv
|
|
26
|
+
|
|
27
|
+
from ._registry import PROVIDERS
|
|
28
|
+
|
|
29
|
+
load_dotenv() # pulls keys from .env in cwd
|
|
30
|
+
|
|
31
|
+
# HTTP statuses that mean "this key/provider is unusable — skip the whole provider"
|
|
32
|
+
_SKIP_PROVIDER_STATUSES = {401, 403}
|
|
33
|
+
# Any other >=400 is treated as retryable: try next model, then next provider.
|
|
34
|
+
|
|
35
|
+
_TIMEOUT = 60 # seconds per request
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class AllProvidersFailed(Exception):
|
|
39
|
+
"""Raised when every configured model on every provider failed."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class _SkipProvider(Exception):
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class _TryNextModel(Exception):
|
|
47
|
+
pass
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _call(provider, model, messages, **params):
|
|
51
|
+
"""Single OpenAI-compatible request. Returns assistant text or raises."""
|
|
52
|
+
key = os.getenv(provider["key_env"])
|
|
53
|
+
if not key:
|
|
54
|
+
raise _SkipProvider(f"no key ({provider['key_env']}) in env")
|
|
55
|
+
|
|
56
|
+
resp = requests.post(
|
|
57
|
+
f"{provider['base_url']}/chat/completions",
|
|
58
|
+
headers={
|
|
59
|
+
"Authorization": f"Bearer {key}",
|
|
60
|
+
"Content-Type": "application/json",
|
|
61
|
+
},
|
|
62
|
+
json={"model": model, "messages": messages, **params},
|
|
63
|
+
timeout=_TIMEOUT,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
if resp.status_code in _SKIP_PROVIDER_STATUSES:
|
|
67
|
+
raise _SkipProvider(f"HTTP {resp.status_code}: {resp.text[:200]}")
|
|
68
|
+
if resp.status_code >= 400:
|
|
69
|
+
# retryable at the model level (bad model name, rate limit, 5xx, ...)
|
|
70
|
+
raise _TryNextModel(f"HTTP {resp.status_code}: {resp.text[:200]}")
|
|
71
|
+
|
|
72
|
+
return resp.json()["choices"][0]["message"]["content"]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def chat(prompt=None, messages=None, return_meta=False, verbose=False, **params):
|
|
76
|
+
"""
|
|
77
|
+
Send a chat request, failing over across models then providers.
|
|
78
|
+
|
|
79
|
+
Pass either `prompt` (str) or `messages` (OpenAI-format list). Extra kwargs
|
|
80
|
+
(temperature, max_tokens, ...) are forwarded to the API verbatim.
|
|
81
|
+
"""
|
|
82
|
+
if messages is None:
|
|
83
|
+
if prompt is None:
|
|
84
|
+
raise ValueError("provide either `prompt` or `messages`")
|
|
85
|
+
messages = [{"role": "user", "content": prompt}]
|
|
86
|
+
|
|
87
|
+
errors = [] # (provider, model, reason) for diagnostics
|
|
88
|
+
|
|
89
|
+
for provider in PROVIDERS:
|
|
90
|
+
for model in provider["models"]:
|
|
91
|
+
try:
|
|
92
|
+
text = _call(provider, model, messages, **params)
|
|
93
|
+
if verbose:
|
|
94
|
+
print(f"[ok] {provider['name']} / {model}")
|
|
95
|
+
if return_meta:
|
|
96
|
+
return text, {"provider": provider["name"], "model": model}
|
|
97
|
+
return text
|
|
98
|
+
except _SkipProvider as e:
|
|
99
|
+
if verbose:
|
|
100
|
+
print(f"[skip provider] {provider['name']}: {e}")
|
|
101
|
+
errors.append((provider["name"], "*", str(e)))
|
|
102
|
+
break # stop trying this provider's other models
|
|
103
|
+
except _TryNextModel as e:
|
|
104
|
+
if verbose:
|
|
105
|
+
print(f"[next model] {provider['name']} / {model}: {e}")
|
|
106
|
+
errors.append((provider["name"], model, str(e)))
|
|
107
|
+
continue
|
|
108
|
+
except (requests.Timeout, requests.ConnectionError) as e:
|
|
109
|
+
if verbose:
|
|
110
|
+
print(f"[network] {provider['name']} / {model}: {e}")
|
|
111
|
+
errors.append((provider["name"], model, f"network: {e}"))
|
|
112
|
+
continue
|
|
113
|
+
|
|
114
|
+
detail = "\n".join(f" {p}/{m}: {r}" for p, m, r in errors) or " (no keys found)"
|
|
115
|
+
raise AllProvidersFailed(f"All providers/models failed:\n{detail}")
|