tollfree 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,11 @@
1
+ .env
2
+ .venv/
3
+ __pycache__/
4
+
5
+ # build artifacts
6
+ dist/
7
+ build/
8
+ *.egg-info/
9
+
10
+ # node
11
+ node_modules/
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.5
2
+ Name: tollfree
3
+ Version: 0.1.0
4
+ Summary: One chat() call over free-tier LLM providers, with automatic model + provider failover.
5
+ Project-URL: Homepage, https://github.com/luarss/free-llm-router
6
+ Project-URL: Source, https://github.com/luarss/free-llm-router
7
+ Author-email: luarss <song.luar@a5x.ai>
8
+ License: MIT
9
+ Keywords: failover,free-tier,groq,llm,openai,router
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Requires-Python: >=3.9
14
+ Requires-Dist: python-dotenv>=1.0
15
+ Requires-Dist: requests>=2.31
16
+ Provides-Extra: test
17
+ Requires-Dist: pytest>=8.0; extra == 'test'
18
+ Description-Content-Type: text/markdown
19
+
20
+ # tollfree (Python)
21
+
22
+ One `chat()` call over free-tier LLM providers, with automatic failover:
23
+ **rotate a provider's models first, then fall over to the next provider.**
24
+
25
+ Keys are read straight from `.env`. Missing keys are skipped automatically — you
26
+ only need one to start.
27
+
28
+ ## Install
29
+
30
+ ```bash
31
+ pip install tollfree # or: uv add tollfree
32
+ ```
33
+
34
+ ## Use
35
+
36
+ ```python
37
+ from tollfree import chat
38
+
39
+ reply = chat("Explain the CAP theorem in two sentences.")
40
+
41
+ # with metadata about which provider/model actually answered:
42
+ reply, meta = chat("hi", return_meta=True, temperature=0.2)
43
+ # meta -> {'provider': 'google', 'model': 'gemini-3.6-flash'}
44
+ ```
45
+
46
+ CLI:
47
+
48
+ ```bash
49
+ tollfree "your prompt here" # chat
50
+ tollfree probe # verify keys + limits, spends no completion tokens
51
+ ```
52
+
53
+ ## Failover logic
54
+
55
+ ```
56
+ for provider in PROVIDERS: # priority order (providers.json)
57
+ for model in provider.models: # default = first
58
+ try -> return on success
59
+ 401/403 -> skip whole provider (bad key)
60
+ 429/404/5xx/timeout -> next model, then next provider
61
+ raise AllProvidersFailed # only when everything is exhausted
62
+ ```
63
+
64
+ ## Configuration
65
+
66
+ Set keys in `.env` (or the environment):
67
+
68
+ ```
69
+ GROQ_API_KEY=
70
+ CEREBRAS_API_KEY=
71
+ GEMINI_API_KEY=
72
+ OPENROUTER_API_KEY=
73
+ MISTRAL_API_KEY=
74
+ ZAI_API_KEY=
75
+ ```
76
+
77
+ The provider/model registry is `providers.json`, kept in sync from the
78
+ [central repo](https://github.com/luarss/free-llm-router) which is the source of
79
+ truth for both the Python and npm packages.
@@ -0,0 +1,60 @@
1
+ # tollfree (Python)
2
+
3
+ One `chat()` call over free-tier LLM providers, with automatic failover:
4
+ **rotate a provider's models first, then fall over to the next provider.**
5
+
6
+ Keys are read straight from `.env`. Missing keys are skipped automatically — you
7
+ only need one to start.
8
+
9
+ ## Install
10
+
11
+ ```bash
12
+ pip install tollfree # or: uv add tollfree
13
+ ```
14
+
15
+ ## Use
16
+
17
+ ```python
18
+ from tollfree import chat
19
+
20
+ reply = chat("Explain the CAP theorem in two sentences.")
21
+
22
+ # with metadata about which provider/model actually answered:
23
+ reply, meta = chat("hi", return_meta=True, temperature=0.2)
24
+ # meta -> {'provider': 'google', 'model': 'gemini-3.6-flash'}
25
+ ```
26
+
27
+ CLI:
28
+
29
+ ```bash
30
+ tollfree "your prompt here" # chat
31
+ tollfree probe # verify keys + limits, spends no completion tokens
32
+ ```
33
+
34
+ ## Failover logic
35
+
36
+ ```
37
+ for provider in PROVIDERS: # priority order (providers.json)
38
+ for model in provider.models: # default = first
39
+ try -> return on success
40
+ 401/403 -> skip whole provider (bad key)
41
+ 429/404/5xx/timeout -> next model, then next provider
42
+ raise AllProvidersFailed # only when everything is exhausted
43
+ ```
44
+
45
+ ## Configuration
46
+
47
+ Set keys in `.env` (or the environment):
48
+
49
+ ```
50
+ GROQ_API_KEY=
51
+ CEREBRAS_API_KEY=
52
+ GEMINI_API_KEY=
53
+ OPENROUTER_API_KEY=
54
+ MISTRAL_API_KEY=
55
+ ZAI_API_KEY=
56
+ ```
57
+
58
+ The provider/model registry is `providers.json`, kept in sync from the
59
+ [central repo](https://github.com/luarss/free-llm-router) which is the source of
60
+ truth for both the Python and npm packages.
@@ -0,0 +1,45 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "tollfree"
7
+ version = "0.1.0"
8
+ description = "One chat() call over free-tier LLM providers, with automatic model + provider failover."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "luarss", email = "song.luar@a5x.ai" }]
13
+ keywords = ["llm", "openai", "groq", "failover", "router", "free-tier"]
14
+ classifiers = [
15
+ "Programming Language :: Python :: 3",
16
+ "License :: OSI Approved :: MIT License",
17
+ "Operating System :: OS Independent",
18
+ ]
19
+ dependencies = [
20
+ "requests>=2.31",
21
+ "python-dotenv>=1.0",
22
+ ]
23
+
24
+ [project.optional-dependencies]
25
+ test = ["pytest>=8.0"]
26
+
27
+ [project.urls]
28
+ Homepage = "https://github.com/luarss/free-llm-router"
29
+ Source = "https://github.com/luarss/free-llm-router"
30
+
31
+ [project.scripts]
32
+ tollfree = "tollfree.cli:main"
33
+
34
+ [tool.hatch.build.targets.wheel]
35
+ packages = ["src/tollfree"]
36
+
37
+ [tool.hatch.build.targets.wheel.force-include]
38
+ "src/tollfree/providers.json" = "tollfree/providers.json"
39
+
40
+ [tool.hatch.build.targets.sdist]
41
+ include = ["src/tollfree", "README.md"]
42
+
43
+ [tool.pytest.ini_options]
44
+ testpaths = ["tests"]
45
+ addopts = "-q"
@@ -0,0 +1,9 @@
1
+ """tollfree — one chat() call over free-tier LLM providers, with automatic
2
+ model + provider failover. See https://github.com/luarss/free-llm-router."""
3
+
4
+ from ._registry import PROVIDERS
5
+ from .probe import probe, probe_all
6
+ from .router import AllProvidersFailed, chat
7
+
8
+ __all__ = ["chat", "probe", "probe_all", "PROVIDERS", "AllProvidersFailed"]
9
+ __version__ = "0.1.0"
@@ -0,0 +1,17 @@
1
+ """Load the canonical provider registry bundled with the package.
2
+
3
+ The registry lives in providers.json (kept in sync from the repo root by
4
+ scripts/sync-providers). Loading it here keeps a single source of truth.
5
+ """
6
+
7
+ import json
8
+ from importlib.resources import files
9
+
10
+
11
+ def load_providers():
12
+ """Return the provider list from the bundled providers.json."""
13
+ raw = files(__package__).joinpath("providers.json").read_text(encoding="utf-8")
14
+ return json.loads(raw)["providers"]
15
+
16
+
17
+ PROVIDERS = load_providers()
@@ -0,0 +1,23 @@
1
+ """Console entry point: `tollfree "prompt"` to chat, `tollfree probe` to check keys."""
2
+
3
+ import sys
4
+
5
+ from .probe import probe_all
6
+ from .router import chat
7
+
8
+
9
+ def main(argv=None):
10
+ argv = list(sys.argv[1:] if argv is None else argv)
11
+
12
+ if argv and argv[0] == "probe":
13
+ probe_all()
14
+ return
15
+
16
+ q = " ".join(argv) or "Say hello in one short sentence."
17
+ answer, meta = chat(q, return_meta=True, verbose=True)
18
+ print(f"\n--- {meta['provider']} / {meta['model']} ---")
19
+ print(answer)
20
+
21
+
22
+ if __name__ == "__main__":
23
+ main()
@@ -0,0 +1,88 @@
1
+ """
2
+ Probe providers WITHOUT spending completion quota.
3
+
4
+ For each provider that has a key in .env, this:
5
+ - GET /models -> verifies the key + auth, lists available models
6
+ (free; does not consume chat/token quota)
7
+ - prints any rate-limit headers the provider returns (real remaining quota)
8
+ - for OpenRouter, also GET /key for credit/limit/usage
9
+ """
10
+
11
+ import os
12
+
13
+ import requests
14
+ from dotenv import load_dotenv
15
+
16
+ from ._registry import PROVIDERS
17
+
18
+ load_dotenv()
19
+
20
+ # Header names providers use to advertise limits/quota. We match loosely.
21
+ _LIMIT_HINTS = ("ratelimit", "rate-limit", "retry-after", "x-request", "quota")
22
+
23
+ _TIMEOUT = 20
24
+
25
+
26
+ def _limit_headers(headers):
27
+ out = {}
28
+ for k, v in headers.items():
29
+ lk = k.lower()
30
+ if any(h in lk for h in _LIMIT_HINTS):
31
+ out[k] = v
32
+ return out
33
+
34
+
35
+ def probe(provider):
36
+ """Probe a single provider dict; prints a status line (and any limit headers)."""
37
+ key = os.getenv(provider["key_env"])
38
+ name = provider["name"]
39
+ if not key:
40
+ print(f"— {name:12s} SKIP (no {provider['key_env']} in .env)")
41
+ return
42
+
43
+ try:
44
+ r = requests.get(
45
+ f"{provider['base_url']}/models",
46
+ headers={"Authorization": f"Bearer {key}"},
47
+ timeout=_TIMEOUT,
48
+ )
49
+ except requests.RequestException as e:
50
+ print(f"✘ {name:12s} NET ({e})")
51
+ return
52
+
53
+ if r.status_code == 200:
54
+ try:
55
+ models = r.json().get("data", [])
56
+ n = len(models)
57
+ except ValueError:
58
+ n = "?"
59
+ print(f"✔ {name:12s} OK key valid, {n} models visible")
60
+ elif r.status_code in (401, 403):
61
+ print(f"✘ {name:12s} AUTH HTTP {r.status_code} — bad/expired key")
62
+ else:
63
+ print(f"⚠ {name:12s} HTTP {r.status_code} — {r.text[:80]}")
64
+
65
+ for hk, hv in _limit_headers(r.headers).items():
66
+ print(f" · {hk}: {hv}")
67
+
68
+ # OpenRouter exposes exact limit/usage/credit for free.
69
+ if name == "openrouter" and r.status_code == 200:
70
+ try:
71
+ kr = requests.get(
72
+ "https://openrouter.ai/api/v1/key",
73
+ headers={"Authorization": f"Bearer {key}"},
74
+ timeout=_TIMEOUT,
75
+ ).json().get("data", {})
76
+ print(
77
+ f" · usage=${kr.get('usage')} limit=${kr.get('limit')} "
78
+ f"free_tier={kr.get('is_free_tier')}"
79
+ )
80
+ except requests.RequestException:
81
+ pass
82
+
83
+
84
+ def probe_all():
85
+ """Probe every provider in the registry."""
86
+ print("Probing providers (no completion tokens spent)\n")
87
+ for p in PROVIDERS:
88
+ probe(p)
@@ -0,0 +1,60 @@
1
+ {
2
+ "_comment": "Canonical provider registry — single source of truth for both the PyPI and npm packages. Priority order matters (top = tried first). Every provider speaks the OpenAI-compatible /chat/completions API, so the only per-provider data is base_url, the env var holding the key, and the model list (first model = default). To add a provider: add a block and set its key env var. Run scripts/sync-providers to copy this into both packages.",
3
+ "providers": [
4
+ {
5
+ "name": "groq",
6
+ "base_url": "https://api.groq.com/openai/v1",
7
+ "key_env": "GROQ_API_KEY",
8
+ "models": [
9
+ "llama-3.3-70b-versatile",
10
+ "llama-3.1-8b-instant",
11
+ "qwen/qwen3-32b"
12
+ ]
13
+ },
14
+ {
15
+ "name": "cerebras",
16
+ "base_url": "https://api.cerebras.ai/v1",
17
+ "key_env": "CEREBRAS_API_KEY",
18
+ "models": [
19
+ "llama-3.3-70b",
20
+ "llama3.1-8b",
21
+ "qwen-3-32b"
22
+ ]
23
+ },
24
+ {
25
+ "name": "google",
26
+ "base_url": "https://generativelanguage.googleapis.com/v1beta/openai",
27
+ "key_env": "GEMINI_API_KEY",
28
+ "models": [
29
+ "gemini-3.6-flash",
30
+ "gemini-3.5-flash-lite"
31
+ ]
32
+ },
33
+ {
34
+ "name": "openrouter",
35
+ "base_url": "https://openrouter.ai/api/v1",
36
+ "key_env": "OPENROUTER_API_KEY",
37
+ "models": [
38
+ "meta-llama/llama-3.3-70b-instruct:free",
39
+ "google/gemini-2.0-flash-exp:free"
40
+ ]
41
+ },
42
+ {
43
+ "name": "mistral",
44
+ "base_url": "https://api.mistral.ai/v1",
45
+ "key_env": "MISTRAL_API_KEY",
46
+ "models": [
47
+ "mistral-small-latest",
48
+ "open-mistral-nemo"
49
+ ]
50
+ },
51
+ {
52
+ "name": "zai",
53
+ "base_url": "https://api.z.ai/api/paas/v4",
54
+ "key_env": "ZAI_API_KEY",
55
+ "models": [
56
+ "glm-4-flash"
57
+ ]
58
+ }
59
+ ]
60
+ }
@@ -0,0 +1,115 @@
1
+ """
2
+ tollfree router: one chat() call, automatic model + provider failover.
3
+
4
+ Rotation order:
5
+ 1. Within a provider, rotate through its models (default first).
6
+ 2. When a provider's models are all exhausted, fall over to the next provider.
7
+
8
+ Usage:
9
+ from tollfree import chat
10
+ reply = chat("Explain CAP theorem in two sentences.")
11
+ print(reply)
12
+
13
+ # or with full message history / options
14
+ reply, meta = chat(
15
+ messages=[{"role": "user", "content": "hi"}],
16
+ temperature=0.2,
17
+ return_meta=True,
18
+ )
19
+ print(meta) # {'provider': 'groq', 'model': 'llama-3.3-70b-versatile'}
20
+ """
21
+
22
+ import os
23
+
24
+ import requests
25
+ from dotenv import load_dotenv
26
+
27
+ from ._registry import PROVIDERS
28
+
29
+ load_dotenv() # pulls keys from .env in cwd
30
+
31
+ # HTTP statuses that mean "this key/provider is unusable — skip the whole provider"
32
+ _SKIP_PROVIDER_STATUSES = {401, 403}
33
+ # Any other >=400 is treated as retryable: try next model, then next provider.
34
+
35
+ _TIMEOUT = 60 # seconds per request
36
+
37
+
38
+ class AllProvidersFailed(Exception):
39
+ """Raised when every configured model on every provider failed."""
40
+
41
+
42
+ class _SkipProvider(Exception):
43
+ pass
44
+
45
+
46
+ class _TryNextModel(Exception):
47
+ pass
48
+
49
+
50
+ def _call(provider, model, messages, **params):
51
+ """Single OpenAI-compatible request. Returns assistant text or raises."""
52
+ key = os.getenv(provider["key_env"])
53
+ if not key:
54
+ raise _SkipProvider(f"no key ({provider['key_env']}) in env")
55
+
56
+ resp = requests.post(
57
+ f"{provider['base_url']}/chat/completions",
58
+ headers={
59
+ "Authorization": f"Bearer {key}",
60
+ "Content-Type": "application/json",
61
+ },
62
+ json={"model": model, "messages": messages, **params},
63
+ timeout=_TIMEOUT,
64
+ )
65
+
66
+ if resp.status_code in _SKIP_PROVIDER_STATUSES:
67
+ raise _SkipProvider(f"HTTP {resp.status_code}: {resp.text[:200]}")
68
+ if resp.status_code >= 400:
69
+ # retryable at the model level (bad model name, rate limit, 5xx, ...)
70
+ raise _TryNextModel(f"HTTP {resp.status_code}: {resp.text[:200]}")
71
+
72
+ return resp.json()["choices"][0]["message"]["content"]
73
+
74
+
75
+ def chat(prompt=None, messages=None, return_meta=False, verbose=False, **params):
76
+ """
77
+ Send a chat request, failing over across models then providers.
78
+
79
+ Pass either `prompt` (str) or `messages` (OpenAI-format list). Extra kwargs
80
+ (temperature, max_tokens, ...) are forwarded to the API verbatim.
81
+ """
82
+ if messages is None:
83
+ if prompt is None:
84
+ raise ValueError("provide either `prompt` or `messages`")
85
+ messages = [{"role": "user", "content": prompt}]
86
+
87
+ errors = [] # (provider, model, reason) for diagnostics
88
+
89
+ for provider in PROVIDERS:
90
+ for model in provider["models"]:
91
+ try:
92
+ text = _call(provider, model, messages, **params)
93
+ if verbose:
94
+ print(f"[ok] {provider['name']} / {model}")
95
+ if return_meta:
96
+ return text, {"provider": provider["name"], "model": model}
97
+ return text
98
+ except _SkipProvider as e:
99
+ if verbose:
100
+ print(f"[skip provider] {provider['name']}: {e}")
101
+ errors.append((provider["name"], "*", str(e)))
102
+ break # stop trying this provider's other models
103
+ except _TryNextModel as e:
104
+ if verbose:
105
+ print(f"[next model] {provider['name']} / {model}: {e}")
106
+ errors.append((provider["name"], model, str(e)))
107
+ continue
108
+ except (requests.Timeout, requests.ConnectionError) as e:
109
+ if verbose:
110
+ print(f"[network] {provider['name']} / {model}: {e}")
111
+ errors.append((provider["name"], model, f"network: {e}"))
112
+ continue
113
+
114
+ detail = "\n".join(f" {p}/{m}: {r}" for p, m, r in errors) or " (no keys found)"
115
+ raise AllProvidersFailed(f"All providers/models failed:\n{detail}")