llm-proxy-cli 0.5.1__tar.gz → 0.5.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/PKG-INFO +2 -2
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/README.md +1 -1
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/PKG-INFO +2 -2
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.py +19 -11
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/pyproject.toml +1 -1
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/tests/test_circuit.py +43 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/tests/test_query_ai.py +35 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/LICENSE +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/SOURCES.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/dependency_links.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/entry_points.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/requires.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/llm_proxy_cli.egg-info/top_level.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/setup.cfg +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.3}/tests/test_discovery.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-proxy-cli
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.3
|
|
4
4
|
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
5
|
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -29,7 +29,7 @@ Dynamic: license-file
|
|
|
29
29
|
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
30
30
|
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
31
31
|
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
32
|
-
- **Reasoning Extraction**: Automatically extracts
|
|
32
|
+
- **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
|
|
33
33
|
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
34
34
|
|
|
35
35
|
## 🏗️ Architecture & Under the Hood
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
14
14
|
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
15
15
|
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
16
|
-
- **Reasoning Extraction**: Automatically extracts
|
|
16
|
+
- **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
|
|
17
17
|
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
18
18
|
|
|
19
19
|
## 🏗️ Architecture & Under the Hood
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-proxy-cli
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.3
|
|
4
4
|
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
5
|
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -29,7 +29,7 @@ Dynamic: license-file
|
|
|
29
29
|
- **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
|
|
30
30
|
- **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
|
|
31
31
|
- **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
|
|
32
|
-
- **Reasoning Extraction**: Automatically extracts
|
|
32
|
+
- **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
|
|
33
33
|
- **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
|
|
34
34
|
|
|
35
35
|
## 🏗️ Architecture & Under the Hood
|
|
@@ -9,7 +9,7 @@ import logging
|
|
|
9
9
|
from openai import OpenAI
|
|
10
10
|
from filelock import FileLock, Timeout
|
|
11
11
|
|
|
12
|
-
__version__ = "0.5.
|
|
12
|
+
__version__ = "0.5.3"
|
|
13
13
|
|
|
14
14
|
# Optional import for anthropic
|
|
15
15
|
try:
|
|
@@ -43,14 +43,15 @@ AUTO_DISCOVERY_TIMEOUT = 5 # seconds - keep the "auto" resolve snappy
|
|
|
43
43
|
|
|
44
44
|
def get_api_key(provider):
|
|
45
45
|
keys_file = os.path.join(get_config_dir(), "keys.json")
|
|
46
|
+
keys = {}
|
|
46
47
|
if os.path.exists(keys_file):
|
|
47
48
|
try:
|
|
48
49
|
with open(keys_file, 'r', encoding='utf-8') as f:
|
|
49
50
|
keys = json.load(f)
|
|
50
|
-
env_name = f"{provider.upper()}_API_KEY"
|
|
51
51
|
except Exception:
|
|
52
52
|
pass
|
|
53
|
-
|
|
53
|
+
env_name = f"{provider.upper()}_API_KEY"
|
|
54
|
+
return os.environ.get(env_name) or keys.get(env_name)
|
|
54
55
|
|
|
55
56
|
|
|
56
57
|
PROVIDERS = {
|
|
@@ -336,8 +337,16 @@ class CircuitBreaker:
|
|
|
336
337
|
with FileLock(self.lock_file, timeout=5):
|
|
337
338
|
circuit = self.load()
|
|
338
339
|
if model_id not in circuit:
|
|
339
|
-
circuit[model_id] = {'failures': 0, 'cooldown_until': 0}
|
|
340
|
+
circuit[model_id] = {'failures': 0, 'cooldown_until': 0, 'last_failure_time': 0}
|
|
341
|
+
|
|
342
|
+
# Decay old failures after 300 seconds
|
|
343
|
+
last_fail = circuit[model_id].get('last_failure_time', 0)
|
|
344
|
+
if time.time() - last_fail > 300:
|
|
345
|
+
circuit[model_id]['failures'] = 0
|
|
346
|
+
|
|
340
347
|
circuit[model_id]['failures'] += 1
|
|
348
|
+
circuit[model_id]['last_failure_time'] = time.time()
|
|
349
|
+
|
|
341
350
|
if circuit[model_id]['failures'] >= self.max_failures:
|
|
342
351
|
circuit[model_id]['cooldown_until'] = time.time() + self.cooldown_seconds
|
|
343
352
|
circuit[model_id]['failures'] = 0
|
|
@@ -417,7 +426,6 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
417
426
|
) as stream:
|
|
418
427
|
for text in stream.text_stream:
|
|
419
428
|
if stream_out:
|
|
420
|
-
import sys
|
|
421
429
|
sys.stdout.write(text)
|
|
422
430
|
sys.stdout.flush()
|
|
423
431
|
full_content += text
|
|
@@ -436,13 +444,11 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
436
444
|
if reasoning:
|
|
437
445
|
full_reasoning += reasoning
|
|
438
446
|
if stream_out:
|
|
439
|
-
import sys
|
|
440
447
|
sys.stderr.write(reasoning)
|
|
441
448
|
sys.stderr.flush()
|
|
442
449
|
content = chunk.choices[0].delta.content
|
|
443
450
|
if content:
|
|
444
451
|
if stream_out:
|
|
445
|
-
import sys
|
|
446
452
|
sys.stdout.write(content)
|
|
447
453
|
sys.stdout.flush()
|
|
448
454
|
full_content += content
|
|
@@ -456,6 +462,9 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
456
462
|
except Exception as e:
|
|
457
463
|
error_msg = str(e).lower()
|
|
458
464
|
logger.error(f"Attempt {attempt+1} failed for {resolved_key}: {str(e)}")
|
|
465
|
+
if stream_out and (full_content or full_reasoning):
|
|
466
|
+
sys.stderr.write(f"\n[STREAM INTERRUPTED: {str(e)} - FALLBACK TRIGGERED]\n")
|
|
467
|
+
sys.stderr.flush()
|
|
459
468
|
status_code = getattr(e, "status_code", None)
|
|
460
469
|
if status_code in (404, 401, 403) or "404" in error_msg or "not found" in error_msg or "auth" in error_msg:
|
|
461
470
|
break
|
|
@@ -474,7 +483,7 @@ def main():
|
|
|
474
483
|
if hasattr(sys.stdout, 'reconfigure'):
|
|
475
484
|
sys.stdout.reconfigure(encoding='utf-8')
|
|
476
485
|
|
|
477
|
-
parser = argparse.ArgumentParser(description="
|
|
486
|
+
parser = argparse.ArgumentParser(description="LLM Proxy CLI: A fault-tolerant CLI tool for LLM delegation.")
|
|
478
487
|
parser.add_argument("-v", "--version", action="version", version=f"LLM Proxy CLI v{__version__}")
|
|
479
488
|
parser.add_argument("-m", "--models", required=True, help="Comma-separated list of provider:model fallbacks (e.g. nvidia:nemotron,groq:llama3, or groq:auto-smart / groq:auto-fast).")
|
|
480
489
|
parser.add_argument("-p", "--prompt", help="The prompt text to send to the model.")
|
|
@@ -506,9 +515,8 @@ def main():
|
|
|
506
515
|
|
|
507
516
|
cb = CircuitBreaker(args.project, args.max_failures, args.cooldown)
|
|
508
517
|
try:
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
print(response)
|
|
518
|
+
query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
|
|
519
|
+
print()
|
|
512
520
|
except Exception as e:
|
|
513
521
|
logger.error(str(e))
|
|
514
522
|
sys.exit(1)
|
|
@@ -46,3 +46,46 @@ def test_circuit_breaker_success_reset(tmp_path):
|
|
|
46
46
|
cb.record_success("modelC")
|
|
47
47
|
circuit = cb.load()
|
|
48
48
|
assert "modelC" not in circuit or circuit["modelC"]["failures"] == 0
|
|
49
|
+
|
|
50
|
+
import time
|
|
51
|
+
|
|
52
|
+
def test_circuit_breaker_decay():
|
|
53
|
+
cb = CircuitBreaker("decay_test", max_failures=2, cooldown_seconds=60)
|
|
54
|
+
|
|
55
|
+
# Clean state
|
|
56
|
+
if os.path.exists(cb.circuit_file):
|
|
57
|
+
os.remove(cb.circuit_file)
|
|
58
|
+
|
|
59
|
+
import llm_proxy_cli
|
|
60
|
+
|
|
61
|
+
# Mock time.time to simulate passage of time
|
|
62
|
+
original_time = time.time
|
|
63
|
+
current_mock_time = original_time()
|
|
64
|
+
|
|
65
|
+
def mock_time():
|
|
66
|
+
return current_mock_time
|
|
67
|
+
|
|
68
|
+
llm_proxy_cli.time.time = mock_time
|
|
69
|
+
|
|
70
|
+
try:
|
|
71
|
+
# Failure 1 at T=0
|
|
72
|
+
cb.record_failure("test:model")
|
|
73
|
+
assert cb.check_health("test:model") == True
|
|
74
|
+
|
|
75
|
+
# Advance time by 400 seconds (past 300s decay threshold)
|
|
76
|
+
current_mock_time += 400
|
|
77
|
+
|
|
78
|
+
# Failure 2 at T=400
|
|
79
|
+
# This should reset the counter to 0 before adding 1, so it won't trip (max_failures=2)
|
|
80
|
+
cb.record_failure("test:model")
|
|
81
|
+
assert cb.check_health("test:model") == True
|
|
82
|
+
|
|
83
|
+
# Failure 3 immediately after (T=400)
|
|
84
|
+
# Counter becomes 2 -> trips!
|
|
85
|
+
cb.record_failure("test:model")
|
|
86
|
+
assert cb.check_health("test:model") == False
|
|
87
|
+
|
|
88
|
+
finally:
|
|
89
|
+
llm_proxy_cli.time.time = original_time
|
|
90
|
+
if os.path.exists(cb.circuit_file):
|
|
91
|
+
os.remove(cb.circuit_file)
|
|
@@ -197,3 +197,38 @@ def test_query_ai_resolves_auto_alias_before_calling_provider(monkeypatch, mock_
|
|
|
197
197
|
# circuit breaker must key on the resolved model, never on the literal "auto-smart" alias
|
|
198
198
|
mock_cb.check_health.assert_called_with("groq:resolved-model-xyz")
|
|
199
199
|
mock_cb.record_success.assert_called_once_with("groq:resolved-model-xyz")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def test_query_ai_streaming_output(capsys, monkeypatch, mock_cb):
|
|
203
|
+
"""Test that stream_out=True sends reasoning to stderr and content to stdout, exactly once."""
|
|
204
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": "dummy", "api_key": "key"})
|
|
205
|
+
|
|
206
|
+
mock_client = MagicMock()
|
|
207
|
+
mock_chunk1 = MagicMock()
|
|
208
|
+
mock_chunk1.choices = [MagicMock()]
|
|
209
|
+
mock_chunk1.choices[0].delta.reasoning_content = "thinking..."
|
|
210
|
+
mock_chunk1.choices[0].delta.content = ""
|
|
211
|
+
|
|
212
|
+
mock_chunk2 = MagicMock()
|
|
213
|
+
mock_chunk2.choices = [MagicMock()]
|
|
214
|
+
mock_chunk2.choices[0].delta.reasoning_content = ""
|
|
215
|
+
mock_chunk2.choices[0].delta.content = "hello world"
|
|
216
|
+
|
|
217
|
+
mock_client.chat.completions.create.return_value = [mock_chunk1, mock_chunk2]
|
|
218
|
+
|
|
219
|
+
factory = openai_factory({"dummy": mock_client})
|
|
220
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
221
|
+
response = router.query_ai("groq:llama", "hi", mock_cb, stream_out=True)
|
|
222
|
+
|
|
223
|
+
captured = capsys.readouterr()
|
|
224
|
+
|
|
225
|
+
# 1. Stdout must only contain "hello world" (no reasoning)
|
|
226
|
+
assert "hello world" in captured.out
|
|
227
|
+
assert "thinking..." not in captured.out
|
|
228
|
+
|
|
229
|
+
# 2. Stderr must contain the reasoning
|
|
230
|
+
assert "thinking..." in captured.err
|
|
231
|
+
|
|
232
|
+
# 3. The returned string still contains everything for library usage
|
|
233
|
+
assert "hello world" in response
|
|
234
|
+
assert "thinking..." in response
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|