llm-proxy-cli 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_proxy_cli-0.5.0/LICENSE +21 -0
- llm_proxy_cli-0.5.0/PKG-INFO +117 -0
- llm_proxy_cli-0.5.0/README.md +101 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/PKG-INFO +117 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/SOURCES.txt +13 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/dependency_links.txt +1 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/entry_points.txt +2 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/requires.txt +3 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.egg-info/top_level.txt +1 -0
- llm_proxy_cli-0.5.0/llm_proxy_cli.py +495 -0
- llm_proxy_cli-0.5.0/pyproject.toml +28 -0
- llm_proxy_cli-0.5.0/setup.cfg +4 -0
- llm_proxy_cli-0.5.0/tests/test_circuit.py +48 -0
- llm_proxy_cli-0.5.0/tests/test_discovery.py +61 -0
- llm_proxy_cli-0.5.0/tests/test_query_ai.py +199 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kerem Barbaros Karnabat
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: llm-proxy-cli
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
|
+
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Requires-Python: >=3.8
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: openai>=1.0.0
|
|
13
|
+
Requires-Dist: filelock>=3.12.0
|
|
14
|
+
Requires-Dist: anthropic>=0.30.0
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# LLM Proxy CLI
|
|
18
|
+
|
|
19
|
+

|
|
20
|
+

|
|
21
|
+

|
|
22
|
+
|
|
23
|
+
> A lightweight, fault-tolerant CLI tool for delegating LLM tasks to expert models across multiple providers (Nvidia NIM, Groq, OpenAI, Anthropic Claude, Gemini).
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
## ⚡ Features
|
|
27
|
+
- **Multi-Provider Support**: Seamlessly route requests to `nvidia`, `groq`, `openai`, `anthropic`, or `gemini`.
|
|
28
|
+
- **Dynamic Model Discovery**: Never hardcode a model name again. Use `auto-smart` or `auto-fast` and the router will auto-select the best model.
|
|
29
|
+
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
30
|
+
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
31
|
+
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
32
|
+
- **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
|
|
33
|
+
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
34
|
+
|
|
35
|
+
## 🏗️ Architecture & Under the Hood
|
|
36
|
+
- **Language**: Python 3
|
|
37
|
+
- **Libraries**: openai, filelock, anthropic
|
|
38
|
+
- **Design Pattern**: Circuit Breaker, Chain of Responsibility (Fallback Routing), and Dynamic Caching.
|
|
39
|
+
|
|
40
|
+
The router uses a `FileLock`-backed JSON state (`circuit_breaker.json`) to track failures across concurrent runs.
|
|
41
|
+
If an endpoint times out or returns a 5xx error more than `MAX_FAILURES` times, the circuit trips and forces the router to skip that endpoint for the next 120 seconds, immediately trying the next fallback model.
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
## 📦 Installation
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
# Install via pip
|
|
48
|
+
pip install llm-proxy-cli
|
|
49
|
+
|
|
50
|
+
# Or for local development:
|
|
51
|
+
# git clone https://github.com/cadakerem/llm-proxy-cli.git
|
|
52
|
+
# cd llm-proxy-cli
|
|
53
|
+
# pip install -e .
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## 🔑 Configuration & API Keys
|
|
57
|
+
|
|
58
|
+
The router looks for API keys in your environment variables or in ~/.config/llm-proxy-cli/keys.json.
|
|
59
|
+
|
|
60
|
+
Supported environment variables:
|
|
61
|
+
- NVIDIA_API_KEY
|
|
62
|
+
- GROQ_API_KEY
|
|
63
|
+
- OPENAI_API_KEY
|
|
64
|
+
- ANTHROPIC_API_KEY
|
|
65
|
+
- GEMINI_API_KEY
|
|
66
|
+
|
|
67
|
+
## 💻 Usage
|
|
68
|
+
|
|
69
|
+
The tool is designed to **automatically** discover and use the best model without you having to memorize model names (using `auto-smart` and `auto-fast`).
|
|
70
|
+
|
|
71
|
+
### 1. Automatic Model Selection (Recommended)
|
|
72
|
+
Instead of guessing which model is currently the best or active on the API, simply use `auto-smart` (for complex coding/reasoning tasks) or `auto-fast` (for quick tasks).
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
# Auto-select the smartest model on Nvidia (e.g., Nemotron or Llama 3.1 405B)
|
|
76
|
+
llm-proxy-cli -m "nvidia:auto-smart" -p "Write a React button."
|
|
77
|
+
|
|
78
|
+
# Auto-select the fastest model on Groq
|
|
79
|
+
llm-proxy-cli -m "groq:auto-fast" -p "Summarize this text."
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
### 2. Chained Automatic Fallback
|
|
83
|
+
If Nvidia goes down or hits a rate limit, you can instantly fall back to Groq's best model by separating them with a comma:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
llm-proxy-cli -m "nvidia:auto-smart,groq:auto-smart" -p "Refactor this python script."
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
### 3. Specific / Manual Model Selection
|
|
90
|
+
If you have a specific model you want to use, you can still hardcode it directly:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
# Use Laguna, and fallback to a specific Groq model if it fails
|
|
94
|
+
llm-proxy-cli -m "nvidia:poolside/laguna-xs-2.1,groq:groq/compound" -p "Explain quantum entanglement."
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### ⚠️ Troubleshooting & Known Quirks: Nvidia EULA (404 Not Found)
|
|
98
|
+
Nvidia NIM requires users to manually accept the **End User License Agreement (EULA)** for certain models on their website before using them via API. If you haven't accepted the EULA for a dynamically discovered model, Nvidia returns a cryptic `404 Not Found` error.
|
|
99
|
+
The Smart Router intercepts this behavior automatically and will print a clear warning.
|
|
100
|
+
|
|
101
|
+
**How to Fix:**
|
|
102
|
+
1. Log into the [Nvidia Build Portal](https://build.nvidia.com).
|
|
103
|
+
2. Search for the exact model name shown in the warning and click to run a quick test prompt to accept the terms.
|
|
104
|
+
3. Or bypass auto-discovery completely by explicitly hardcoding a model:
|
|
105
|
+
```bash
|
|
106
|
+
llm-proxy-cli -m "nvidia:meta/llama-3.2-11b-vision-instruct" -p "Hello"
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## 🧑💻 Developer & Contributions
|
|
110
|
+
Developed by Kerem Barbaros Karnabat ([@cadakerem](https://github.com/cadakerem)).
|
|
111
|
+
|
|
112
|
+
> **Note on Repository Structure:** The core routing logic, dynamic model discovery, and circuit breaker patterns are entirely contained within `llm_proxy_cli.py` to ensure maximum portability. Unit tests are located in the `tests/` directory, and `SKILL.md` provides instructions for integrating this tool as a native AI agent skill.
|
|
113
|
+
|
|
114
|
+
Contributions, issues, and feature requests are welcome! Feel free to check the [Issues page](../../issues).
|
|
115
|
+
|
|
116
|
+
## 📜 License
|
|
117
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# LLM Proxy CLI
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+

|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
> A lightweight, fault-tolerant CLI tool for delegating LLM tasks to expert models across multiple providers (Nvidia NIM, Groq, OpenAI, Anthropic Claude, Gemini).
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
## ⚡ Features
|
|
11
|
+
- **Multi-Provider Support**: Seamlessly route requests to `nvidia`, `groq`, `openai`, `anthropic`, or `gemini`.
|
|
12
|
+
- **Dynamic Model Discovery**: Never hardcode a model name again. Use `auto-smart` or `auto-fast` and the router will auto-select the best model.
|
|
13
|
+
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
14
|
+
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
15
|
+
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
16
|
+
- **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
|
|
17
|
+
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
18
|
+
|
|
19
|
+
## 🏗️ Architecture & Under the Hood
|
|
20
|
+
- **Language**: Python 3
|
|
21
|
+
- **Libraries**: openai, filelock, anthropic
|
|
22
|
+
- **Design Pattern**: Circuit Breaker, Chain of Responsibility (Fallback Routing), and Dynamic Caching.
|
|
23
|
+
|
|
24
|
+
The router uses a `FileLock`-backed JSON state (`circuit_breaker.json`) to track failures across concurrent runs.
|
|
25
|
+
If an endpoint times out or returns a 5xx error more than `MAX_FAILURES` times, the circuit trips and forces the router to skip that endpoint for the next 120 seconds, immediately trying the next fallback model.
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
## 📦 Installation
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
# Install via pip
|
|
32
|
+
pip install llm-proxy-cli
|
|
33
|
+
|
|
34
|
+
# Or for local development:
|
|
35
|
+
# git clone https://github.com/cadakerem/llm-proxy-cli.git
|
|
36
|
+
# cd llm-proxy-cli
|
|
37
|
+
# pip install -e .
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## 🔑 Configuration & API Keys
|
|
41
|
+
|
|
42
|
+
The router looks for API keys in your environment variables or in ~/.config/llm-proxy-cli/keys.json.
|
|
43
|
+
|
|
44
|
+
Supported environment variables:
|
|
45
|
+
- NVIDIA_API_KEY
|
|
46
|
+
- GROQ_API_KEY
|
|
47
|
+
- OPENAI_API_KEY
|
|
48
|
+
- ANTHROPIC_API_KEY
|
|
49
|
+
- GEMINI_API_KEY
|
|
50
|
+
|
|
51
|
+
## 💻 Usage
|
|
52
|
+
|
|
53
|
+
The tool is designed to **automatically** discover and use the best model without you having to memorize model names (using `auto-smart` and `auto-fast`).
|
|
54
|
+
|
|
55
|
+
### 1. Automatic Model Selection (Recommended)
|
|
56
|
+
Instead of guessing which model is currently the best or active on the API, simply use `auto-smart` (for complex coding/reasoning tasks) or `auto-fast` (for quick tasks).
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
# Auto-select the smartest model on Nvidia (e.g., Nemotron or Llama 3.1 405B)
|
|
60
|
+
llm-proxy-cli -m "nvidia:auto-smart" -p "Write a React button."
|
|
61
|
+
|
|
62
|
+
# Auto-select the fastest model on Groq
|
|
63
|
+
llm-proxy-cli -m "groq:auto-fast" -p "Summarize this text."
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### 2. Chained Automatic Fallback
|
|
67
|
+
If Nvidia goes down or hits a rate limit, you can instantly fall back to Groq's best model by separating them with a comma:
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
llm-proxy-cli -m "nvidia:auto-smart,groq:auto-smart" -p "Refactor this python script."
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### 3. Specific / Manual Model Selection
|
|
74
|
+
If you have a specific model you want to use, you can still hardcode it directly:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
# Use Laguna, and fallback to a specific Groq model if it fails
|
|
78
|
+
llm-proxy-cli -m "nvidia:poolside/laguna-xs-2.1,groq:groq/compound" -p "Explain quantum entanglement."
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
### ⚠️ Troubleshooting & Known Quirks: Nvidia EULA (404 Not Found)
|
|
82
|
+
Nvidia NIM requires users to manually accept the **End User License Agreement (EULA)** for certain models on their website before using them via API. If you haven't accepted the EULA for a dynamically discovered model, Nvidia returns a cryptic `404 Not Found` error.
|
|
83
|
+
The Smart Router intercepts this behavior automatically and will print a clear warning.
|
|
84
|
+
|
|
85
|
+
**How to Fix:**
|
|
86
|
+
1. Log into the [Nvidia Build Portal](https://build.nvidia.com).
|
|
87
|
+
2. Search for the exact model name shown in the warning and click to run a quick test prompt to accept the terms.
|
|
88
|
+
3. Or bypass auto-discovery completely by explicitly hardcoding a model:
|
|
89
|
+
```bash
|
|
90
|
+
llm-proxy-cli -m "nvidia:meta/llama-3.2-11b-vision-instruct" -p "Hello"
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## 🧑💻 Developer & Contributions
|
|
94
|
+
Developed by Kerem Barbaros Karnabat ([@cadakerem](https://github.com/cadakerem)).
|
|
95
|
+
|
|
96
|
+
> **Note on Repository Structure:** The core routing logic, dynamic model discovery, and circuit breaker patterns are entirely contained within `llm_proxy_cli.py` to ensure maximum portability. Unit tests are located in the `tests/` directory, and `SKILL.md` provides instructions for integrating this tool as a native AI agent skill.
|
|
97
|
+
|
|
98
|
+
Contributions, issues, and feature requests are welcome! Feel free to check the [Issues page](../../issues).
|
|
99
|
+
|
|
100
|
+
## 📜 License
|
|
101
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: llm-proxy-cli
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
|
+
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Requires-Python: >=3.8
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: openai>=1.0.0
|
|
13
|
+
Requires-Dist: filelock>=3.12.0
|
|
14
|
+
Requires-Dist: anthropic>=0.30.0
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# LLM Proxy CLI
|
|
18
|
+
|
|
19
|
+

|
|
20
|
+

|
|
21
|
+

|
|
22
|
+
|
|
23
|
+
> A lightweight, fault-tolerant CLI tool for delegating LLM tasks to expert models across multiple providers (Nvidia NIM, Groq, OpenAI, Anthropic Claude, Gemini).
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
## ⚡ Features
|
|
27
|
+
- **Multi-Provider Support**: Seamlessly route requests to `nvidia`, `groq`, `openai`, `anthropic`, or `gemini`.
|
|
28
|
+
- **Dynamic Model Discovery**: Never hardcode a model name again. Use `auto-smart` or `auto-fast` and the router will auto-select the best model.
|
|
29
|
+
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
30
|
+
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
31
|
+
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
32
|
+
- **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
|
|
33
|
+
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
34
|
+
|
|
35
|
+
## 🏗️ Architecture & Under the Hood
|
|
36
|
+
- **Language**: Python 3
|
|
37
|
+
- **Libraries**: openai, filelock, anthropic
|
|
38
|
+
- **Design Pattern**: Circuit Breaker, Chain of Responsibility (Fallback Routing), and Dynamic Caching.
|
|
39
|
+
|
|
40
|
+
The router uses a `FileLock`-backed JSON state (`circuit_breaker.json`) to track failures across concurrent runs.
|
|
41
|
+
If an endpoint times out or returns a 5xx error more than `MAX_FAILURES` times, the circuit trips and forces the router to skip that endpoint for the next 120 seconds, immediately trying the next fallback model.
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
## 📦 Installation
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
# Install via pip
|
|
48
|
+
pip install llm-proxy-cli
|
|
49
|
+
|
|
50
|
+
# Or for local development:
|
|
51
|
+
# git clone https://github.com/cadakerem/llm-proxy-cli.git
|
|
52
|
+
# cd llm-proxy-cli
|
|
53
|
+
# pip install -e .
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## 🔑 Configuration & API Keys
|
|
57
|
+
|
|
58
|
+
The router looks for API keys in your environment variables or in ~/.config/llm-proxy-cli/keys.json.
|
|
59
|
+
|
|
60
|
+
Supported environment variables:
|
|
61
|
+
- NVIDIA_API_KEY
|
|
62
|
+
- GROQ_API_KEY
|
|
63
|
+
- OPENAI_API_KEY
|
|
64
|
+
- ANTHROPIC_API_KEY
|
|
65
|
+
- GEMINI_API_KEY
|
|
66
|
+
|
|
67
|
+
## 💻 Usage
|
|
68
|
+
|
|
69
|
+
The tool is designed to **automatically** discover and use the best model without you having to memorize model names (using `auto-smart` and `auto-fast`).
|
|
70
|
+
|
|
71
|
+
### 1. Automatic Model Selection (Recommended)
|
|
72
|
+
Instead of guessing which model is currently the best or active on the API, simply use `auto-smart` (for complex coding/reasoning tasks) or `auto-fast` (for quick tasks).
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
# Auto-select the smartest model on Nvidia (e.g., Nemotron or Llama 3.1 405B)
|
|
76
|
+
llm-proxy-cli -m "nvidia:auto-smart" -p "Write a React button."
|
|
77
|
+
|
|
78
|
+
# Auto-select the fastest model on Groq
|
|
79
|
+
llm-proxy-cli -m "groq:auto-fast" -p "Summarize this text."
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
### 2. Chained Automatic Fallback
|
|
83
|
+
If Nvidia goes down or hits a rate limit, you can instantly fall back to Groq's best model by separating them with a comma:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
llm-proxy-cli -m "nvidia:auto-smart,groq:auto-smart" -p "Refactor this python script."
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
### 3. Specific / Manual Model Selection
|
|
90
|
+
If you have a specific model you want to use, you can still hardcode it directly:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
# Use Laguna, and fallback to a specific Groq model if it fails
|
|
94
|
+
llm-proxy-cli -m "nvidia:poolside/laguna-xs-2.1,groq:groq/compound" -p "Explain quantum entanglement."
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### ⚠️ Troubleshooting & Known Quirks: Nvidia EULA (404 Not Found)
|
|
98
|
+
Nvidia NIM requires users to manually accept the **End User License Agreement (EULA)** for certain models on their website before using them via API. If you haven't accepted the EULA for a dynamically discovered model, Nvidia returns a cryptic `404 Not Found` error.
|
|
99
|
+
The Smart Router intercepts this behavior automatically and will print a clear warning.
|
|
100
|
+
|
|
101
|
+
**How to Fix:**
|
|
102
|
+
1. Log into the [Nvidia Build Portal](https://build.nvidia.com).
|
|
103
|
+
2. Search for the exact model name shown in the warning and click to run a quick test prompt to accept the terms.
|
|
104
|
+
3. Or bypass auto-discovery completely by explicitly hardcoding a model:
|
|
105
|
+
```bash
|
|
106
|
+
llm-proxy-cli -m "nvidia:meta/llama-3.2-11b-vision-instruct" -p "Hello"
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## 🧑💻 Developer & Contributions
|
|
110
|
+
Developed by Kerem Barbaros Karnabat ([@cadakerem](https://github.com/cadakerem)).
|
|
111
|
+
|
|
112
|
+
> **Note on Repository Structure:** The core routing logic, dynamic model discovery, and circuit breaker patterns are entirely contained within `llm_proxy_cli.py` to ensure maximum portability. Unit tests are located in the `tests/` directory, and `SKILL.md` provides instructions for integrating this tool as a native AI agent skill.
|
|
113
|
+
|
|
114
|
+
Contributions, issues, and feature requests are welcome! Feel free to check the [Issues page](../../issues).
|
|
115
|
+
|
|
116
|
+
## 📜 License
|
|
117
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
llm_proxy_cli.py
|
|
4
|
+
pyproject.toml
|
|
5
|
+
llm_proxy_cli.egg-info/PKG-INFO
|
|
6
|
+
llm_proxy_cli.egg-info/SOURCES.txt
|
|
7
|
+
llm_proxy_cli.egg-info/dependency_links.txt
|
|
8
|
+
llm_proxy_cli.egg-info/entry_points.txt
|
|
9
|
+
llm_proxy_cli.egg-info/requires.txt
|
|
10
|
+
llm_proxy_cli.egg-info/top_level.txt
|
|
11
|
+
tests/test_circuit.py
|
|
12
|
+
tests/test_discovery.py
|
|
13
|
+
tests/test_query_ai.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
llm_proxy_cli
|
|
@@ -0,0 +1,495 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import os
|
|
3
|
+
import json
|
|
4
|
+
import time
|
|
5
|
+
import random
|
|
6
|
+
import re
|
|
7
|
+
import argparse
|
|
8
|
+
import logging
|
|
9
|
+
from openai import OpenAI
|
|
10
|
+
from filelock import FileLock, Timeout
|
|
11
|
+
|
|
12
|
+
__version__ = "0.5.0"
|
|
13
|
+
|
|
14
|
+
# Optional import for anthropic
|
|
15
|
+
try:
|
|
16
|
+
import anthropic
|
|
17
|
+
HAS_ANTHROPIC = True
|
|
18
|
+
except ImportError:
|
|
19
|
+
HAS_ANTHROPIC = False
|
|
20
|
+
|
|
21
|
+
# Setup Logging
|
|
22
|
+
logger = logging.getLogger("smart_router")
|
|
23
|
+
handler = logging.StreamHandler(sys.stderr)
|
|
24
|
+
formatter = logging.Formatter("[%(levelname)s] %(message)s")
|
|
25
|
+
handler.setFormatter(formatter)
|
|
26
|
+
logger.addHandler(handler)
|
|
27
|
+
logger.setLevel(logging.INFO)
|
|
28
|
+
|
|
29
|
+
def get_config_dir():
|
|
30
|
+
d = os.path.expanduser("~/.config/llm-proxy-cli")
|
|
31
|
+
os.makedirs(d, exist_ok=True)
|
|
32
|
+
return d
|
|
33
|
+
|
|
34
|
+
def get_cache_dir():
|
|
35
|
+
d = os.path.expanduser("~/.cache/llm-proxy-cli")
|
|
36
|
+
os.makedirs(d, exist_ok=True)
|
|
37
|
+
return d
|
|
38
|
+
|
|
39
|
+
# --- Auto model discovery config ---
|
|
40
|
+
AUTO_CACHE_TTL = 24 * 60 * 60 # refresh discovery at most once per day
|
|
41
|
+
AUTO_DISCOVERY_TIMEOUT = 5 # seconds - keep the "auto" resolve snappy
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def get_api_key(provider):
|
|
45
|
+
keys_file = os.path.join(get_config_dir(), "keys.json")
|
|
46
|
+
if os.path.exists(keys_file):
|
|
47
|
+
try:
|
|
48
|
+
with open(keys_file, 'r', encoding='utf-8') as f:
|
|
49
|
+
keys = json.load(f)
|
|
50
|
+
env_name = f"{provider.upper()}_API_KEY"
|
|
51
|
+
if keys.get(env_name):
|
|
52
|
+
return keys[env_name]
|
|
53
|
+
except Exception:
|
|
54
|
+
pass
|
|
55
|
+
return os.environ.get(f"{provider.upper()}_API_KEY")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
PROVIDERS = {
|
|
59
|
+
"nvidia": {"base_url": "https://integrate.api.nvidia.com/v1", "api_key": get_api_key("NVIDIA")},
|
|
60
|
+
"openai": {"base_url": "https://api.openai.com/v1", "api_key": get_api_key("OPENAI")},
|
|
61
|
+
"groq": {"base_url": "https://api.groq.com/openai/v1", "api_key": get_api_key("GROQ")},
|
|
62
|
+
"gemini": {"base_url": "https://generativelanguage.googleapis.com/v1beta/openai/", "api_key": get_api_key("GEMINI")},
|
|
63
|
+
"anthropic": {"api_key": get_api_key("ANTHROPIC")}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ---------------------------------------------------------------------------
|
|
68
|
+
# Dynamic Model Auto-Discovery
|
|
69
|
+
#
|
|
70
|
+
# Usage: pass "provider:auto-smart" or "provider:auto-fast" instead of a
|
|
71
|
+
# hardcoded model name, e.g.
|
|
72
|
+
# llm-proxy-cli -m "groq:auto-smart,nvidia:auto-fast" -p "..."
|
|
73
|
+
# ("auto" and "auto-max" are kept as aliases of "auto-smart" for backwards
|
|
74
|
+
# compatibility with existing scripts/cron jobs.)
|
|
75
|
+
#
|
|
76
|
+
# The router hits the provider's /models endpoint, drops anything that isn't
|
|
77
|
+
# a general-purpose chat/completions model (guardrail filters, moderation,
|
|
78
|
+
# embeddings, TTS/STT, rerankers - these show up in /models but will 400 on
|
|
79
|
+
# a normal chat request), scores what's left, and picks the best one.
|
|
80
|
+
# Results are cached locally per (provider, mode) for AUTO_CACHE_TTL seconds
|
|
81
|
+
# so normal calls never pay the discovery latency.
|
|
82
|
+
# ---------------------------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
# Model ids containing any of these are never valid chat-completion candidates,
|
|
85
|
+
# regardless of mode - excluding them up front is what "auto-fast" needs to
|
|
86
|
+
# avoid latching onto a tiny 86M-parameter safety/guard classifier just
|
|
87
|
+
# because it has the smallest size in its name.
|
|
88
|
+
NON_CHAT_KEYWORDS = (
|
|
89
|
+
"guard", "moderation", "safety", "embed", "embedding", "rerank",
|
|
90
|
+
"whisper", "tts", "speech", "audio", "clip", "classifier",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _is_chat_candidate(model_id):
|
|
95
|
+
name = model_id.lower()
|
|
96
|
+
return not any(kw in name for kw in NON_CHAT_KEYWORDS)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _score_model_smart(model_id):
|
|
100
|
+
"""Higher score = bigger / newer / more capable. Used by auto-smart."""
|
|
101
|
+
name = model_id.lower()
|
|
102
|
+
# Strip dates like 2024-08-06 or context sizes like 32768
|
|
103
|
+
name = re.sub(r'20\d{2}[-]?\d{2}[-]?\d{2}', '', name)
|
|
104
|
+
name = re.sub(r'\d{4,}', '', name)
|
|
105
|
+
score = 0.0
|
|
106
|
+
|
|
107
|
+
if "opus" in name: score += 50
|
|
108
|
+
elif "sonnet" in name: score += 40
|
|
109
|
+
elif "haiku" in name: score += 30
|
|
110
|
+
|
|
111
|
+
# Parameter size, e.g. "8b", "70b", "405b" -> biggest wins, weighted heavily.
|
|
112
|
+
size_matches = re.findall(r'(\d+(?:\.\d+)?)\s*b(?!\w)', name)
|
|
113
|
+
if size_matches:
|
|
114
|
+
score += max(float(s) for s in size_matches) * 10
|
|
115
|
+
name_wo_size = re.sub(r'\d+(?:\.\d+)?\s*b(?!\w)', '', name)
|
|
116
|
+
else:
|
|
117
|
+
name_wo_size = name
|
|
118
|
+
|
|
119
|
+
# Any remaining digits are treated as generation/version numbers,
|
|
120
|
+
# e.g. "llama-4", "gemini-2.5", "v3.3" -> higher wins.
|
|
121
|
+
version_matches = re.findall(r'(\d+(?:\.\d+)?)', name_wo_size)
|
|
122
|
+
if version_matches:
|
|
123
|
+
score += max(float(v) for v in version_matches) * 5
|
|
124
|
+
|
|
125
|
+
for kw, bonus in (("instruct", 3), ("versatile", 3), ("reasoning", 4), ("chat", 1)):
|
|
126
|
+
if kw in name:
|
|
127
|
+
score += bonus
|
|
128
|
+
for kw, penalty in (("mini", -2), ("lite", -2), ("tiny", -3), ("preview", -1), ("deprecated", -100)):
|
|
129
|
+
if kw in name:
|
|
130
|
+
score += penalty
|
|
131
|
+
|
|
132
|
+
return score
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _score_model_fast(model_id):
|
|
136
|
+
"""Higher score = smaller / snappier chat model. Used by auto-fast.
|
|
137
|
+
NON_CHAT_KEYWORDS models are filtered out before this ever runs, so
|
|
138
|
+
"smallest wins" can't land on a non-chat model anymore."""
|
|
139
|
+
name = model_id.lower()
|
|
140
|
+
name = re.sub(r'20\d{2}[-]?\d{2}[-]?\d{2}', '', name)
|
|
141
|
+
name = re.sub(r'\d{4,}', '', name)
|
|
142
|
+
score = 0.0
|
|
143
|
+
|
|
144
|
+
if "haiku" in name: score += 20
|
|
145
|
+
elif "sonnet" in name: score += 10
|
|
146
|
+
|
|
147
|
+
size_matches = re.findall(r'(\d+(?:\.\d+)?)\s*b(?!\w)', name)
|
|
148
|
+
if size_matches:
|
|
149
|
+
score -= max(float(s) for s in size_matches) * 10 # smaller size = higher score
|
|
150
|
+
name_wo_size = re.sub(r'\d+(?:\.\d+)?\s*b(?!\w)', '', name)
|
|
151
|
+
else:
|
|
152
|
+
name_wo_size = name
|
|
153
|
+
|
|
154
|
+
version_matches = re.findall(r'(\d+(?:\.\d+)?)', name_wo_size)
|
|
155
|
+
if version_matches:
|
|
156
|
+
score += max(float(v) for v in version_matches) * 5 # still prefer the newer generation
|
|
157
|
+
|
|
158
|
+
for kw, bonus in (("instant", 4), ("flash", 4), ("turbo", 4), ("mini", 3), ("lite", 3), ("small", 2), ("instruct", 1)):
|
|
159
|
+
if kw in name:
|
|
160
|
+
score += bonus
|
|
161
|
+
for kw, penalty in (("preview", -1), ("deprecated", -100)):
|
|
162
|
+
if kw in name:
|
|
163
|
+
score += penalty
|
|
164
|
+
|
|
165
|
+
return score
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_SCORERS = {"smart": _score_model_smart, "fast": _score_model_fast}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _auto_cache_path(provider, mode):
|
|
172
|
+
return os.path.join(get_cache_dir(), f"auto_models_cache_{provider}_{mode}.json")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _load_auto_cache(provider, mode):
|
|
176
|
+
path = _auto_cache_path(provider, mode)
|
|
177
|
+
if os.path.exists(path):
|
|
178
|
+
try:
|
|
179
|
+
with open(path, 'r', encoding='utf-8') as f:
|
|
180
|
+
data = json.load(f)
|
|
181
|
+
if time.time() - data.get('timestamp', 0) < AUTO_CACHE_TTL:
|
|
182
|
+
return data.get('best_model')
|
|
183
|
+
except Exception:
|
|
184
|
+
pass
|
|
185
|
+
return None
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _save_auto_cache(provider, mode, best_model):
|
|
189
|
+
path = _auto_cache_path(provider, mode)
|
|
190
|
+
try:
|
|
191
|
+
with open(path, 'w', encoding='utf-8') as f:
|
|
192
|
+
json.dump({'timestamp': time.time(), 'best_model': best_model}, f)
|
|
193
|
+
except Exception:
|
|
194
|
+
pass
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
VERIFY_TIMEOUT = 8 # seconds per live verification call
|
|
198
|
+
MAX_VERIFY_CANDIDATES = 3 # how many top-scored candidates to actually try before giving up
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _verify_chat_model(provider, provider_config, model_id):
|
|
202
|
+
"""Send one minimal real chat request to confirm this model id actually serves
|
|
203
|
+
chat/completions (not TTS, embeddings, a gated model needing terms acceptance, etc.).
|
|
204
|
+
This is the real safety net - NON_CHAT_KEYWORDS is only a cheap pre-filter to cut
|
|
205
|
+
down how many of these we need to make; the keyword list will always be incomplete
|
|
206
|
+
on its own (see: 'orpheus-v1-english', a TTS model with no matching keyword)."""
|
|
207
|
+
try:
|
|
208
|
+
if provider == "anthropic":
|
|
209
|
+
client = anthropic.Anthropic(api_key=provider_config["api_key"], timeout=VERIFY_TIMEOUT)
|
|
210
|
+
client.messages.create(model=model_id, max_tokens=1, messages=[{"role": "user", "content": "hi"}])
|
|
211
|
+
else:
|
|
212
|
+
client = OpenAI(base_url=provider_config["base_url"], api_key=provider_config["api_key"], timeout=VERIFY_TIMEOUT)
|
|
213
|
+
client.chat.completions.create(model=model_id, messages=[{"role": "user", "content": "hi"}], max_tokens=1)
|
|
214
|
+
return True
|
|
215
|
+
except Exception as e:
|
|
216
|
+
error_str = str(e).lower()
|
|
217
|
+
if provider == "nvidia" and "404" in error_str and "not found for account" in error_str:
|
|
218
|
+
logger.warning(f"Auto-discovery: Nvidia model '{model_id}' requires EULA approval. Please visit https://build.nvidia.com to search and accept the terms for this model.")
|
|
219
|
+
else:
|
|
220
|
+
logger.debug(f"Auto-discovery: '{model_id}' failed live verification: {e}")
|
|
221
|
+
return False
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def discover_best_model(provider, mode="smart", force_refresh=False):
|
|
225
|
+
"""Return the best available chat model id for `provider` under `mode`
|
|
226
|
+
("smart" = biggest/most capable, "fast" = smallest/snappiest), using a
|
|
227
|
+
24h local cache per (provider, mode). Returns None on any failure so
|
|
228
|
+
callers can fall back gracefully.
|
|
229
|
+
|
|
230
|
+
Only runs on a cache miss (once per day per provider/mode), so paying
|
|
231
|
+
a few extra live-verification calls here is worth it for correctness -
|
|
232
|
+
every call after this one comes straight from cache."""
|
|
233
|
+
if mode not in _SCORERS:
|
|
234
|
+
logger.warning(f"Auto-discovery: unknown mode '{mode}', defaulting to 'smart'.")
|
|
235
|
+
mode = "smart"
|
|
236
|
+
|
|
237
|
+
if not force_refresh:
|
|
238
|
+
cached = _load_auto_cache(provider, mode)
|
|
239
|
+
if cached:
|
|
240
|
+
return cached
|
|
241
|
+
|
|
242
|
+
provider_config = PROVIDERS.get(provider)
|
|
243
|
+
if not provider_config or not provider_config.get("api_key"):
|
|
244
|
+
logger.warning(f"Auto-discovery: provider '{provider}' not configured or missing API key.")
|
|
245
|
+
return None
|
|
246
|
+
|
|
247
|
+
try:
|
|
248
|
+
if provider == "anthropic":
|
|
249
|
+
if not HAS_ANTHROPIC:
|
|
250
|
+
return None
|
|
251
|
+
client = anthropic.Anthropic(api_key=provider_config["api_key"], timeout=AUTO_DISCOVERY_TIMEOUT)
|
|
252
|
+
model_ids = [m.id for m in client.models.list().data]
|
|
253
|
+
else:
|
|
254
|
+
client = OpenAI(
|
|
255
|
+
base_url=provider_config["base_url"], api_key=provider_config["api_key"], timeout=AUTO_DISCOVERY_TIMEOUT
|
|
256
|
+
)
|
|
257
|
+
model_ids = [m.id for m in client.models.list().data]
|
|
258
|
+
except Exception as e:
|
|
259
|
+
logger.warning(f"Auto-discovery failed for '{provider}': {e}")
|
|
260
|
+
return None
|
|
261
|
+
|
|
262
|
+
# Cheap pre-filter by name, purely to reduce how many live calls we make below.
|
|
263
|
+
chat_candidates = [m for m in model_ids if _is_chat_candidate(m)]
|
|
264
|
+
if not chat_candidates:
|
|
265
|
+
logger.warning(f"Auto-discovery: no chat-capable models found for '{provider}' (got {len(model_ids)} total, all filtered out by name).")
|
|
266
|
+
return None
|
|
267
|
+
|
|
268
|
+
ranked = sorted(chat_candidates, key=_SCORERS[mode], reverse=True)
|
|
269
|
+
|
|
270
|
+
for candidate in ranked[:MAX_VERIFY_CANDIDATES]:
|
|
271
|
+
if _verify_chat_model(provider, provider_config, candidate):
|
|
272
|
+
logger.info(f"Auto-discovery ({mode}): picked '{candidate}' for '{provider}' (live-verified) out of {len(chat_candidates)}/{len(model_ids)} name-filtered candidates.")
|
|
273
|
+
_save_auto_cache(provider, mode, candidate)
|
|
274
|
+
return candidate
|
|
275
|
+
logger.warning(f"Auto-discovery ({mode}): '{candidate}' failed live verification (gated, non-chat, or otherwise unusable), trying next candidate.")
|
|
276
|
+
|
|
277
|
+
logger.warning(f"Auto-discovery ({mode}): none of the top {min(MAX_VERIFY_CANDIDATES, len(ranked))} name-filtered candidates for '{provider}' passed live verification.")
|
|
278
|
+
return None
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
class CircuitBreaker:
|
|
282
|
+
def __init__(self, project_id, max_failures=2, cooldown_seconds=120):
|
|
283
|
+
self.max_failures = max_failures
|
|
284
|
+
self.cooldown_seconds = cooldown_seconds
|
|
285
|
+
safe_proj = "".join([c if c.isalnum() else "_" for c in project_id])
|
|
286
|
+
self.circuit_file = os.path.join(get_cache_dir(), f"circuit_breaker_{safe_proj}.json")
|
|
287
|
+
self.lock_file = os.path.join(get_cache_dir(), f"circuit_breaker_{safe_proj}.json.lock")
|
|
288
|
+
|
|
289
|
+
def load(self):
|
|
290
|
+
for _ in range(3):
|
|
291
|
+
if os.path.exists(self.circuit_file):
|
|
292
|
+
try:
|
|
293
|
+
with open(self.circuit_file, 'r', encoding='utf-8') as f:
|
|
294
|
+
return json.load(f)
|
|
295
|
+
except Exception:
|
|
296
|
+
time.sleep(random.uniform(0.01, 0.05))
|
|
297
|
+
return {}
|
|
298
|
+
|
|
299
|
+
def save(self, data):
|
|
300
|
+
for _ in range(3):
|
|
301
|
+
try:
|
|
302
|
+
current_time = time.time()
|
|
303
|
+
active_data = {
|
|
304
|
+
k: v for k, v in data.items()
|
|
305
|
+
if v.get('failures', 0) > 0 or v.get('cooldown_until', 0) > current_time
|
|
306
|
+
}
|
|
307
|
+
if not active_data:
|
|
308
|
+
if os.path.exists(self.circuit_file):
|
|
309
|
+
os.remove(self.circuit_file)
|
|
310
|
+
return
|
|
311
|
+
with open(self.circuit_file, 'w', encoding='utf-8') as f:
|
|
312
|
+
json.dump(active_data, f)
|
|
313
|
+
break
|
|
314
|
+
except Exception:
|
|
315
|
+
time.sleep(random.uniform(0.01, 0.05))
|
|
316
|
+
|
|
317
|
+
def check_health(self, model_id):
|
|
318
|
+
try:
|
|
319
|
+
with FileLock(self.lock_file, timeout=5):
|
|
320
|
+
circuit = self.load()
|
|
321
|
+
if model_id in circuit:
|
|
322
|
+
stats = circuit[model_id]
|
|
323
|
+
if stats.get('cooldown_until', 0) > time.time():
|
|
324
|
+
return False
|
|
325
|
+
return True
|
|
326
|
+
except Timeout:
|
|
327
|
+
logger.debug(f"Timeout acquiring lock for {model_id} health check. Assuming healthy.")
|
|
328
|
+
return True
|
|
329
|
+
|
|
330
|
+
def record_failure(self, model_id):
|
|
331
|
+
try:
|
|
332
|
+
with FileLock(self.lock_file, timeout=5):
|
|
333
|
+
circuit = self.load()
|
|
334
|
+
if model_id not in circuit:
|
|
335
|
+
circuit[model_id] = {'failures': 0, 'cooldown_until': 0}
|
|
336
|
+
circuit[model_id]['failures'] += 1
|
|
337
|
+
if circuit[model_id]['failures'] >= self.max_failures:
|
|
338
|
+
circuit[model_id]['cooldown_until'] = time.time() + self.cooldown_seconds
|
|
339
|
+
circuit[model_id]['failures'] = 0
|
|
340
|
+
logger.warning(f"CIRCUIT BREAKER: {model_id} tripped! Cooldown: {self.cooldown_seconds}s.")
|
|
341
|
+
self.save(circuit)
|
|
342
|
+
except Timeout:
|
|
343
|
+
logger.debug(f"Timeout acquiring lock. Could not record failure for {model_id}.")
|
|
344
|
+
|
|
345
|
+
def record_success(self, model_id):
|
|
346
|
+
try:
|
|
347
|
+
with FileLock(self.lock_file, timeout=5):
|
|
348
|
+
circuit = self.load()
|
|
349
|
+
if model_id in circuit and circuit[model_id]['failures'] > 0:
|
|
350
|
+
circuit[model_id]['failures'] = 0
|
|
351
|
+
self.save(circuit)
|
|
352
|
+
except Timeout:
|
|
353
|
+
pass
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
# "auto" / "auto-max" are kept as aliases of "auto-smart" for backwards compatibility.
|
|
357
|
+
_AUTO_ALIASES = {"auto": "smart", "auto-max": "smart", "auto-smart": "smart", "auto-fast": "fast"}
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def parse_model(model_string, force_refresh_auto=False):
|
|
361
|
+
if ":" in model_string:
|
|
362
|
+
provider, model = model_string.split(":", 1)
|
|
363
|
+
provider, model = provider.strip().lower(), model.strip()
|
|
364
|
+
else:
|
|
365
|
+
provider, model = "nvidia", model_string.strip()
|
|
366
|
+
|
|
367
|
+
mode = _AUTO_ALIASES.get(model.lower())
|
|
368
|
+
if mode:
|
|
369
|
+
resolved = discover_best_model(provider, mode=mode, force_refresh=force_refresh_auto)
|
|
370
|
+
if resolved:
|
|
371
|
+
return provider, resolved
|
|
372
|
+
logger.error(f"Auto-discovery unavailable for '{provider}' ({mode}) and no static fallback was given.")
|
|
373
|
+
return provider, model # will fail fast in query_ai (model not found / no key)
|
|
374
|
+
|
|
375
|
+
return provider, model
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeout=30, force_refresh_auto=False):
|
|
379
|
+
if isinstance(models_list, str):
|
|
380
|
+
models_list = [m.strip() for m in models_list.split(',')]
|
|
381
|
+
|
|
382
|
+
for current_model_str in models_list:
|
|
383
|
+
provider, current_model = parse_model(current_model_str, force_refresh_auto=force_refresh_auto)
|
|
384
|
+
# Circuit breaker tracks the *resolved* model, not the literal "auto" alias,
|
|
385
|
+
# since "auto" can point at a different real model over time.
|
|
386
|
+
resolved_key = f"{provider}:{current_model}"
|
|
387
|
+
|
|
388
|
+
if not cb.check_health(resolved_key):
|
|
389
|
+
logger.info(f"Health Check Failed: {resolved_key} is in cooldown. Skipping...")
|
|
390
|
+
continue
|
|
391
|
+
|
|
392
|
+
provider_config = PROVIDERS.get(provider)
|
|
393
|
+
if not provider_config or not provider_config.get("api_key"):
|
|
394
|
+
logger.error(f"Provider '{provider}' not configured or missing API key. Skipping...")
|
|
395
|
+
continue
|
|
396
|
+
|
|
397
|
+
is_nemotron = "nemotron" in current_model.lower()
|
|
398
|
+
model_timeout = 90 if is_nemotron else base_timeout
|
|
399
|
+
|
|
400
|
+
for attempt in range(max_retries):
|
|
401
|
+
try:
|
|
402
|
+
full_reasoning = ""
|
|
403
|
+
full_content = ""
|
|
404
|
+
if provider == "anthropic":
|
|
405
|
+
if not HAS_ANTHROPIC:
|
|
406
|
+
raise ImportError("Anthropic package is missing. 'pip install anthropic' required.")
|
|
407
|
+
client = anthropic.Anthropic(api_key=provider_config["api_key"], timeout=model_timeout, max_retries=0)
|
|
408
|
+
with client.messages.stream(
|
|
409
|
+
model=current_model, max_tokens=4096,
|
|
410
|
+
messages=[{"role": "user", "content": prompt}], temperature=0.7
|
|
411
|
+
) as stream:
|
|
412
|
+
for text in stream.text_stream:
|
|
413
|
+
full_content += text
|
|
414
|
+
else:
|
|
415
|
+
client = OpenAI(
|
|
416
|
+
base_url=provider_config["base_url"], api_key=provider_config["api_key"], timeout=model_timeout, max_retries=0
|
|
417
|
+
)
|
|
418
|
+
extra_body = {"chat_template_kwargs": {"enable_thinking": True}} if (provider == "nvidia" and "nemotron" in current_model.lower()) else {}
|
|
419
|
+
completion = client.chat.completions.create(
|
|
420
|
+
model=current_model, messages=[{"role": "user", "content": prompt}],
|
|
421
|
+
temperature=0.7, max_tokens=4096, extra_body=extra_body if extra_body else None, stream=True
|
|
422
|
+
)
|
|
423
|
+
for chunk in completion:
|
|
424
|
+
if not chunk.choices: continue
|
|
425
|
+
reasoning = getattr(chunk.choices[0].delta, "reasoning_content", None)
|
|
426
|
+
if reasoning: full_reasoning += reasoning
|
|
427
|
+
content = chunk.choices[0].delta.content
|
|
428
|
+
if content: full_content += content
|
|
429
|
+
|
|
430
|
+
cb.record_success(resolved_key)
|
|
431
|
+
output = ""
|
|
432
|
+
if full_reasoning: output += f"--- REASONING ({resolved_key}) ---\n{full_reasoning}\n--- END REASONING ---\n\n"
|
|
433
|
+
output += full_content
|
|
434
|
+
return output
|
|
435
|
+
|
|
436
|
+
except Exception as e:
|
|
437
|
+
error_msg = str(e).lower()
|
|
438
|
+
logger.error(f"Attempt {attempt+1} failed for {resolved_key}: {str(e)}")
|
|
439
|
+
cb.record_failure(resolved_key)
|
|
440
|
+
|
|
441
|
+
status_code = getattr(e, "status_code", None)
|
|
442
|
+
if status_code in (429, 404, 401, 403) or "404" in error_msg or "not found" in error_msg or "auth" in error_msg:
|
|
443
|
+
break
|
|
444
|
+
|
|
445
|
+
if attempt == max_retries - 1: break
|
|
446
|
+
time.sleep((2 ** attempt) + random.uniform(0.1, 1.5))
|
|
447
|
+
|
|
448
|
+
raise RuntimeError("All fallback models failed, timed out, or are in cooldown.")
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def main():
|
|
452
|
+
if hasattr(sys.stdout, 'reconfigure'):
|
|
453
|
+
sys.stdout.reconfigure(encoding='utf-8')
|
|
454
|
+
|
|
455
|
+
parser = argparse.ArgumentParser(description="Smart Router: A fault-tolerant CLI tool for LLM delegation.")
|
|
456
|
+
parser.add_argument("-v", "--version", action="version", version=f"Smart Router v{__version__}")
|
|
457
|
+
parser.add_argument("-m", "--models", required=True, help="Comma-separated list of provider:model fallbacks (e.g. nvidia:nemotron,groq:llama3, or groq:auto-smart / groq:auto-fast).")
|
|
458
|
+
parser.add_argument("-p", "--prompt", help="The prompt text to send to the model.")
|
|
459
|
+
parser.add_argument("-f", "--file", help="Path to a text file containing the prompt.")
|
|
460
|
+
parser.add_argument("--project", default="default", help="Project ID for isolating circuit breaker state.")
|
|
461
|
+
parser.add_argument("--max-failures", type=int, default=2, help="Failures before tripping the circuit breaker.")
|
|
462
|
+
parser.add_argument("--cooldown", type=int, default=120, help="Cooldown in seconds when circuit is tripped.")
|
|
463
|
+
parser.add_argument("--refresh-models", action="store_true", help="Ignore the 24h auto-discovery cache and re-query provider model lists now.")
|
|
464
|
+
args = parser.parse_args()
|
|
465
|
+
|
|
466
|
+
prompt_text = ""
|
|
467
|
+
if args.prompt:
|
|
468
|
+
prompt_text = args.prompt
|
|
469
|
+
elif args.file:
|
|
470
|
+
try:
|
|
471
|
+
with open(args.file, "r", encoding="utf-8") as f:
|
|
472
|
+
prompt_text = f.read()
|
|
473
|
+
except Exception as e:
|
|
474
|
+
logger.error(f"Error reading file: {e}")
|
|
475
|
+
sys.exit(1)
|
|
476
|
+
elif not sys.stdin.isatty():
|
|
477
|
+
prompt_text = sys.stdin.read()
|
|
478
|
+
else:
|
|
479
|
+
parser.error("You must provide a prompt via -p, -f, or stdin (piped input).")
|
|
480
|
+
|
|
481
|
+
if not prompt_text.strip():
|
|
482
|
+
logger.error("Prompt cannot be empty.")
|
|
483
|
+
sys.exit(1)
|
|
484
|
+
|
|
485
|
+
cb = CircuitBreaker(args.project, args.max_failures, args.cooldown)
|
|
486
|
+
try:
|
|
487
|
+
response = query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models)
|
|
488
|
+
print(response)
|
|
489
|
+
except Exception as e:
|
|
490
|
+
logger.error(str(e))
|
|
491
|
+
sys.exit(1)
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
if __name__ == "__main__":
|
|
495
|
+
main()
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "llm-proxy-cli"
|
|
7
|
+
version = "0.5.0"
|
|
8
|
+
authors = [
|
|
9
|
+
{ name="Kerem Barbaros Karnabat", email="kbarbaros@hotmail.com" }
|
|
10
|
+
]
|
|
11
|
+
description = "A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers."
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
requires-python = ">=3.8"
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
]
|
|
19
|
+
dependencies = [
|
|
20
|
+
"openai>=1.0.0",
|
|
21
|
+
"filelock>=3.12.0",
|
|
22
|
+
"anthropic>=0.30.0"
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
[project.scripts]
|
|
26
|
+
llm-proxy-cli = "llm_proxy_cli:main"
|
|
27
|
+
|
|
28
|
+
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import time
|
|
3
|
+
import pytest
|
|
4
|
+
from llm_proxy_cli import CircuitBreaker
|
|
5
|
+
|
|
6
|
+
def test_circuit_breaker_isolation(tmp_path):
|
|
7
|
+
# Test that different projects don't share state
|
|
8
|
+
cb1 = CircuitBreaker("proj1", max_failures=2, cooldown_seconds=10)
|
|
9
|
+
cb2 = CircuitBreaker("proj2", max_failures=2, cooldown_seconds=10)
|
|
10
|
+
|
|
11
|
+
cb1.circuit_file = str(tmp_path / "cb1.json")
|
|
12
|
+
cb1.lock_file = str(tmp_path / "cb1.json.lock")
|
|
13
|
+
cb2.circuit_file = str(tmp_path / "cb2.json")
|
|
14
|
+
cb2.lock_file = str(tmp_path / "cb2.json.lock")
|
|
15
|
+
|
|
16
|
+
cb1.record_failure("modelA")
|
|
17
|
+
cb1.record_failure("modelA")
|
|
18
|
+
|
|
19
|
+
# cb1 should be tripped
|
|
20
|
+
assert cb1.check_health("modelA") == False
|
|
21
|
+
# cb2 should be unaffected
|
|
22
|
+
assert cb2.check_health("modelA") == True
|
|
23
|
+
|
|
24
|
+
def test_circuit_breaker_cooldown(tmp_path):
|
|
25
|
+
cb = CircuitBreaker("test", max_failures=1, cooldown_seconds=1)
|
|
26
|
+
cb.circuit_file = str(tmp_path / "cb.json")
|
|
27
|
+
cb.lock_file = str(tmp_path / "cb.json.lock")
|
|
28
|
+
|
|
29
|
+
assert cb.check_health("modelB") == True
|
|
30
|
+
cb.record_failure("modelB")
|
|
31
|
+
|
|
32
|
+
assert cb.check_health("modelB") == False
|
|
33
|
+
time.sleep(1.1)
|
|
34
|
+
# After cooldown, should be healthy again
|
|
35
|
+
assert cb.check_health("modelB") == True
|
|
36
|
+
|
|
37
|
+
def test_circuit_breaker_success_reset(tmp_path):
|
|
38
|
+
cb = CircuitBreaker("test2", max_failures=2, cooldown_seconds=10)
|
|
39
|
+
cb.circuit_file = str(tmp_path / "cb.json")
|
|
40
|
+
cb.lock_file = str(tmp_path / "cb.json.lock")
|
|
41
|
+
|
|
42
|
+
cb.record_failure("modelC")
|
|
43
|
+
circuit = cb.load()
|
|
44
|
+
assert circuit["modelC"]["failures"] == 1
|
|
45
|
+
|
|
46
|
+
cb.record_success("modelC")
|
|
47
|
+
circuit = cb.load()
|
|
48
|
+
assert "modelC" not in circuit or circuit["modelC"]["failures"] == 0
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Unit tests for the auto-discovery scoring/filtering logic in llm_proxy_cli.py.
|
|
3
|
+
These are all pure functions - no network calls, no API keys needed - so they
|
|
4
|
+
run in CI the same way test_circuit.py does. The live verification call itself
|
|
5
|
+
(_verify_chat_model) is intentionally NOT unit tested here since it requires a
|
|
6
|
+
real API round trip; that's what the --refresh-models manual smoke test is for.
|
|
7
|
+
"""
|
|
8
|
+
import llm_proxy_cli as router
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_is_chat_candidate_filters_known_non_chat_models():
|
|
12
|
+
assert not router._is_chat_candidate("meta-llama/llama-prompt-guard-2-86m")
|
|
13
|
+
assert not router._is_chat_candidate("text-embedding-3-large")
|
|
14
|
+
assert not router._is_chat_candidate("whisper-large-v3")
|
|
15
|
+
assert not router._is_chat_candidate("llama-guard-4-12b")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def test_is_chat_candidate_keeps_ordinary_chat_models():
|
|
19
|
+
assert router._is_chat_candidate("llama-3.3-70b-versatile")
|
|
20
|
+
assert router._is_chat_candidate("gpt-oss-120b")
|
|
21
|
+
assert router._is_chat_candidate("gemma2-9b-it")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_is_chat_candidate_does_not_catch_everything():
|
|
25
|
+
# Documents the known gap this filter has: a non-chat model whose name gives
|
|
26
|
+
# no hint (e.g. Orpheus, a TTS model) will pass the name filter and must be
|
|
27
|
+
# caught by live verification instead. This is why verification exists.
|
|
28
|
+
assert router._is_chat_candidate("canopylabs/orpheus-v1-english")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_score_smart_prefers_bigger_models():
|
|
32
|
+
small = router._score_model_smart("llama-3.1-8b-instant")
|
|
33
|
+
big = router._score_model_smart("llama-3.1-405b-instruct")
|
|
34
|
+
assert big > small
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_score_smart_prefers_newer_generation_at_similar_size():
|
|
38
|
+
older = router._score_model_smart("llama-3.1-70b-versatile")
|
|
39
|
+
newer = router._score_model_smart("llama-4-70b-versatile")
|
|
40
|
+
assert newer > older
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def test_score_fast_prefers_smaller_models():
|
|
44
|
+
small = router._score_model_smart("llama-3.1-8b-instant")
|
|
45
|
+
big = router._score_model_smart("llama-3.1-405b-instruct")
|
|
46
|
+
assert router._score_model_fast("llama-3.1-8b-instant") > router._score_model_fast("llama-3.1-405b-instruct")
|
|
47
|
+
# sanity: smart and fast should disagree on which of these two is "better"
|
|
48
|
+
assert (big > small) != (
|
|
49
|
+
router._score_model_fast("llama-3.1-405b-instruct") > router._score_model_fast("llama-3.1-8b-instant")
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_score_fast_rewards_speed_keywords():
|
|
54
|
+
plain = router._score_model_fast("llama-3.1-8b")
|
|
55
|
+
instant = router._score_model_fast("llama-3.1-8b-instant")
|
|
56
|
+
assert instant > plain
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_discover_best_model_returns_none_without_api_key(monkeypatch):
|
|
60
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": "https://api.groq.com/openai/v1", "api_key": None})
|
|
61
|
+
assert router.discover_best_model("groq", mode="smart") is None
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Integration-style tests for query_ai(): fallback between models, retry behavior,
|
|
3
|
+
and its interaction with CircuitBreaker - all with the network mocked out via
|
|
4
|
+
unittest.mock, so these run in CI with no API keys and no real HTTP calls.
|
|
5
|
+
|
|
6
|
+
test_circuit.py already covers CircuitBreaker's own file-lock/open-close logic
|
|
7
|
+
in isolation, so here the CircuitBreaker is a MagicMock: we're testing query_ai's
|
|
8
|
+
control flow (which model gets tried, when it moves on, when it retries), not
|
|
9
|
+
the breaker's persistence.
|
|
10
|
+
|
|
11
|
+
IMPORTANT: query_ai() constructs a fresh OpenAI(...) client on every retry
|
|
12
|
+
attempt, not once per model. Mocking the OpenAI constructor with a plain
|
|
13
|
+
side_effect=[client_a, client_b] list is a trap - a retried attempt silently
|
|
14
|
+
consumes the *next* list entry (meant for the fallback model) and the test
|
|
15
|
+
passes for the wrong reason. openai_factory() below keys the returned mock
|
|
16
|
+
client by base_url instead, so it's stable across any number of retries.
|
|
17
|
+
"""
|
|
18
|
+
from types import SimpleNamespace
|
|
19
|
+
from unittest.mock import MagicMock, patch
|
|
20
|
+
|
|
21
|
+
import pytest
|
|
22
|
+
|
|
23
|
+
import llm_proxy_cli as router
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def make_openai_chunk(content=None, reasoning=None):
|
|
27
|
+
"""Fake streaming chunk shaped like the OpenAI SDK's ChatCompletionChunk - just
|
|
28
|
+
enough for query_ai's `chunk.choices[0].delta.content` / .reasoning_content access."""
|
|
29
|
+
delta = SimpleNamespace(content=content)
|
|
30
|
+
if reasoning is not None:
|
|
31
|
+
delta.reasoning_content = reasoning
|
|
32
|
+
return SimpleNamespace(choices=[SimpleNamespace(delta=delta)])
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def make_anthropic_stream_cm(text_chunks):
|
|
36
|
+
"""Fake context manager shaped like client.messages.stream(...)."""
|
|
37
|
+
cm = MagicMock()
|
|
38
|
+
cm.__enter__.return_value = SimpleNamespace(text_stream=iter(text_chunks))
|
|
39
|
+
cm.__exit__.return_value = False
|
|
40
|
+
return cm
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def openai_factory(clients_by_base_url):
|
|
44
|
+
"""Fake OpenAI(...) constructor returning a fixed mock client per base_url,
|
|
45
|
+
however many times it's called - see module docstring for why this matters."""
|
|
46
|
+
def _factory(*args, base_url=None, **kwargs):
|
|
47
|
+
return clients_by_base_url[base_url]
|
|
48
|
+
return _factory
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
GROQ_URL = "https://api.groq.com/openai/v1"
|
|
52
|
+
NVIDIA_URL = "https://integrate.api.nvidia.com/v1"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@pytest.fixture(autouse=True)
|
|
56
|
+
def fast_backoff(monkeypatch):
|
|
57
|
+
# Retry backoff sleeps for real seconds otherwise; skip it in tests.
|
|
58
|
+
monkeypatch.setattr(router.time, "sleep", lambda *_: None)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@pytest.fixture
|
|
62
|
+
def mock_cb():
|
|
63
|
+
cb = MagicMock(spec=router.CircuitBreaker)
|
|
64
|
+
cb.check_health.return_value = True
|
|
65
|
+
return cb
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_query_ai_returns_content_on_first_model_success(monkeypatch, mock_cb):
|
|
69
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "test-key"})
|
|
70
|
+
|
|
71
|
+
client = MagicMock()
|
|
72
|
+
client.chat.completions.create.return_value = [
|
|
73
|
+
make_openai_chunk(content="Hello"), make_openai_chunk(content=" world"),
|
|
74
|
+
]
|
|
75
|
+
with patch.object(router, "OpenAI", side_effect=openai_factory({GROQ_URL: client})):
|
|
76
|
+
result = router.query_ai("groq:llama-3.3-70b-versatile", "hi", mock_cb)
|
|
77
|
+
|
|
78
|
+
assert result == "Hello world"
|
|
79
|
+
mock_cb.record_success.assert_called_once_with("groq:llama-3.3-70b-versatile")
|
|
80
|
+
mock_cb.record_failure.assert_not_called()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_query_ai_falls_back_to_next_model_on_failure(monkeypatch, mock_cb):
|
|
84
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "key-a"})
|
|
85
|
+
monkeypatch.setitem(router.PROVIDERS, "nvidia", {"base_url": NVIDIA_URL, "api_key": "key-b"})
|
|
86
|
+
|
|
87
|
+
failing_client = MagicMock()
|
|
88
|
+
failing_client.chat.completions.create.side_effect = Exception("500 Internal Server Error")
|
|
89
|
+
working_client = MagicMock()
|
|
90
|
+
working_client.chat.completions.create.return_value = [make_openai_chunk(content="fallback worked")]
|
|
91
|
+
|
|
92
|
+
factory = openai_factory({GROQ_URL: failing_client, NVIDIA_URL: working_client})
|
|
93
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
94
|
+
result = router.query_ai("groq:llama-3.3-70b-versatile,nvidia:nemotron", "hi", mock_cb, max_retries=2)
|
|
95
|
+
|
|
96
|
+
assert result == "fallback worked"
|
|
97
|
+
# groq must be retried max_retries=2 times (both against the SAME failing client) before moving on
|
|
98
|
+
assert failing_client.chat.completions.create.call_count == 2
|
|
99
|
+
assert mock_cb.record_failure.call_count == 2
|
|
100
|
+
mock_cb.record_failure.assert_called_with("groq:llama-3.3-70b-versatile")
|
|
101
|
+
mock_cb.record_success.assert_called_once_with("nvidia:nemotron")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_query_ai_stops_retrying_immediately_on_404(monkeypatch, mock_cb):
|
|
105
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "key-a"})
|
|
106
|
+
monkeypatch.setitem(router.PROVIDERS, "nvidia", {"base_url": NVIDIA_URL, "api_key": "key-b"})
|
|
107
|
+
|
|
108
|
+
not_found_client = MagicMock()
|
|
109
|
+
not_found_client.chat.completions.create.side_effect = Exception("Error code: 404 - model not found")
|
|
110
|
+
working_client = MagicMock()
|
|
111
|
+
working_client.chat.completions.create.return_value = [make_openai_chunk(content="ok")]
|
|
112
|
+
|
|
113
|
+
factory = openai_factory({GROQ_URL: not_found_client, NVIDIA_URL: working_client})
|
|
114
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
115
|
+
result = router.query_ai("groq:does-not-exist,nvidia:nemotron", "hi", mock_cb, max_retries=3)
|
|
116
|
+
|
|
117
|
+
assert result == "ok"
|
|
118
|
+
# A 404 must break the retry loop after a single attempt, not consume all max_retries=3.
|
|
119
|
+
assert not_found_client.chat.completions.create.call_count == 1
|
|
120
|
+
assert mock_cb.record_failure.call_count == 1
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def test_query_ai_skips_models_in_cooldown(monkeypatch, mock_cb):
|
|
124
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "key-a"})
|
|
125
|
+
monkeypatch.setitem(router.PROVIDERS, "nvidia", {"base_url": NVIDIA_URL, "api_key": "key-b"})
|
|
126
|
+
# groq is in cooldown, nvidia is healthy
|
|
127
|
+
mock_cb.check_health.side_effect = lambda model_id: model_id != "groq:llama-3.3-70b-versatile"
|
|
128
|
+
|
|
129
|
+
working_client = MagicMock()
|
|
130
|
+
working_client.chat.completions.create.return_value = [make_openai_chunk(content="from nvidia")]
|
|
131
|
+
# No entry for GROQ_URL at all - if the code wrongly tried to call groq, this KeyErrors.
|
|
132
|
+
factory = openai_factory({NVIDIA_URL: working_client})
|
|
133
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
134
|
+
result = router.query_ai("groq:llama-3.3-70b-versatile,nvidia:nemotron", "hi", mock_cb)
|
|
135
|
+
|
|
136
|
+
assert result == "from nvidia"
|
|
137
|
+
working_client.chat.completions.create.assert_called_once()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def test_query_ai_skips_provider_without_api_key(monkeypatch, mock_cb):
|
|
141
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": None})
|
|
142
|
+
monkeypatch.setitem(router.PROVIDERS, "nvidia", {"base_url": NVIDIA_URL, "api_key": "key-b"})
|
|
143
|
+
|
|
144
|
+
working_client = MagicMock()
|
|
145
|
+
working_client.chat.completions.create.return_value = [make_openai_chunk(content="ok")]
|
|
146
|
+
# No entry for GROQ_URL - a missing-key provider must never reach the OpenAI constructor.
|
|
147
|
+
factory = openai_factory({NVIDIA_URL: working_client})
|
|
148
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
149
|
+
result = router.query_ai("groq:llama-3.3-70b-versatile,nvidia:nemotron", "hi", mock_cb)
|
|
150
|
+
|
|
151
|
+
assert result == "ok"
|
|
152
|
+
mock_cb.record_failure.assert_not_called()
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def test_query_ai_exits_when_every_model_fails(monkeypatch, mock_cb):
|
|
156
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "key-a"})
|
|
157
|
+
|
|
158
|
+
failing_client = MagicMock()
|
|
159
|
+
failing_client.chat.completions.create.side_effect = Exception("timeout")
|
|
160
|
+
factory = openai_factory({GROQ_URL: failing_client})
|
|
161
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
162
|
+
with pytest.raises(SystemExit) as exc_info:
|
|
163
|
+
router.query_ai("groq:llama-3.3-70b-versatile", "hi", mock_cb, max_retries=2)
|
|
164
|
+
|
|
165
|
+
assert exc_info.value.code == 1
|
|
166
|
+
assert failing_client.chat.completions.create.call_count == 2
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def test_query_ai_uses_anthropic_client_for_anthropic_provider(monkeypatch, mock_cb):
|
|
170
|
+
if not router.HAS_ANTHROPIC:
|
|
171
|
+
pytest.skip("anthropic package not installed")
|
|
172
|
+
monkeypatch.setitem(router.PROVIDERS, "anthropic", {"api_key": "key-c"})
|
|
173
|
+
|
|
174
|
+
mock_client = MagicMock()
|
|
175
|
+
mock_client.messages.stream.return_value = make_anthropic_stream_cm(["Hi", " there"])
|
|
176
|
+
with patch.object(router.anthropic, "Anthropic", return_value=mock_client):
|
|
177
|
+
result = router.query_ai("anthropic:claude-sonnet-4-6", "hi", mock_cb)
|
|
178
|
+
|
|
179
|
+
assert result == "Hi there"
|
|
180
|
+
mock_cb.record_success.assert_called_once_with("anthropic:claude-sonnet-4-6")
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def test_query_ai_resolves_auto_alias_before_calling_provider(monkeypatch, mock_cb):
|
|
184
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": GROQ_URL, "api_key": "key-a"})
|
|
185
|
+
monkeypatch.setattr(router, "discover_best_model", lambda provider, mode="smart", force_refresh=False: "resolved-model-xyz")
|
|
186
|
+
|
|
187
|
+
client = MagicMock()
|
|
188
|
+
client.chat.completions.create.return_value = [make_openai_chunk(content="ok")]
|
|
189
|
+
with patch.object(router, "OpenAI", side_effect=openai_factory({GROQ_URL: client})):
|
|
190
|
+
result = router.query_ai("groq:auto-smart", "hi", mock_cb)
|
|
191
|
+
|
|
192
|
+
assert result == "ok"
|
|
193
|
+
client.chat.completions.create.assert_called_once_with(
|
|
194
|
+
model="resolved-model-xyz", messages=[{"role": "user", "content": "hi"}],
|
|
195
|
+
temperature=0.7, max_tokens=4096, extra_body=None, stream=True,
|
|
196
|
+
)
|
|
197
|
+
# circuit breaker must key on the resolved model, never on the literal "auto-smart" alias
|
|
198
|
+
mock_cb.check_health.assert_called_with("groq:resolved-model-xyz")
|
|
199
|
+
mock_cb.record_success.assert_called_once_with("groq:resolved-model-xyz")
|