llm-proxy-cli 0.5.1__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/PKG-INFO +1 -1
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/PKG-INFO +1 -1
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.py +16 -9
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/pyproject.toml +1 -1
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/tests/test_circuit.py +43 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/tests/test_query_ai.py +35 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/LICENSE +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/README.md +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/SOURCES.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/dependency_links.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/entry_points.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/requires.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/llm_proxy_cli.egg-info/top_level.txt +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/setup.cfg +0 -0
- {llm_proxy_cli-0.5.1 → llm_proxy_cli-0.5.2}/tests/test_discovery.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-proxy-cli
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.2
|
|
4
4
|
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
5
|
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-proxy-cli
|
|
3
|
-
Version: 0.5.
|
|
3
|
+
Version: 0.5.2
|
|
4
4
|
Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
|
|
5
5
|
Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
|
|
6
6
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -9,7 +9,7 @@ import logging
|
|
|
9
9
|
from openai import OpenAI
|
|
10
10
|
from filelock import FileLock, Timeout
|
|
11
11
|
|
|
12
|
-
__version__ = "0.5.
|
|
12
|
+
__version__ = "0.5.2"
|
|
13
13
|
|
|
14
14
|
# Optional import for anthropic
|
|
15
15
|
try:
|
|
@@ -336,8 +336,16 @@ class CircuitBreaker:
|
|
|
336
336
|
with FileLock(self.lock_file, timeout=5):
|
|
337
337
|
circuit = self.load()
|
|
338
338
|
if model_id not in circuit:
|
|
339
|
-
circuit[model_id] = {'failures': 0, 'cooldown_until': 0}
|
|
339
|
+
circuit[model_id] = {'failures': 0, 'cooldown_until': 0, 'last_failure_time': 0}
|
|
340
|
+
|
|
341
|
+
# Decay old failures after 300 seconds
|
|
342
|
+
last_fail = circuit[model_id].get('last_failure_time', 0)
|
|
343
|
+
if time.time() - last_fail > 300:
|
|
344
|
+
circuit[model_id]['failures'] = 0
|
|
345
|
+
|
|
340
346
|
circuit[model_id]['failures'] += 1
|
|
347
|
+
circuit[model_id]['last_failure_time'] = time.time()
|
|
348
|
+
|
|
341
349
|
if circuit[model_id]['failures'] >= self.max_failures:
|
|
342
350
|
circuit[model_id]['cooldown_until'] = time.time() + self.cooldown_seconds
|
|
343
351
|
circuit[model_id]['failures'] = 0
|
|
@@ -417,7 +425,6 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
417
425
|
) as stream:
|
|
418
426
|
for text in stream.text_stream:
|
|
419
427
|
if stream_out:
|
|
420
|
-
import sys
|
|
421
428
|
sys.stdout.write(text)
|
|
422
429
|
sys.stdout.flush()
|
|
423
430
|
full_content += text
|
|
@@ -436,13 +443,11 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
436
443
|
if reasoning:
|
|
437
444
|
full_reasoning += reasoning
|
|
438
445
|
if stream_out:
|
|
439
|
-
import sys
|
|
440
446
|
sys.stderr.write(reasoning)
|
|
441
447
|
sys.stderr.flush()
|
|
442
448
|
content = chunk.choices[0].delta.content
|
|
443
449
|
if content:
|
|
444
450
|
if stream_out:
|
|
445
|
-
import sys
|
|
446
451
|
sys.stdout.write(content)
|
|
447
452
|
sys.stdout.flush()
|
|
448
453
|
full_content += content
|
|
@@ -456,6 +461,9 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
|
|
|
456
461
|
except Exception as e:
|
|
457
462
|
error_msg = str(e).lower()
|
|
458
463
|
logger.error(f"Attempt {attempt+1} failed for {resolved_key}: {str(e)}")
|
|
464
|
+
if stream_out and (full_content or full_reasoning):
|
|
465
|
+
sys.stderr.write(f"\n[STREAM INTERRUPTED: {str(e)} - FALLBACK TRIGGERED]\n")
|
|
466
|
+
sys.stderr.flush()
|
|
459
467
|
status_code = getattr(e, "status_code", None)
|
|
460
468
|
if status_code in (404, 401, 403) or "404" in error_msg or "not found" in error_msg or "auth" in error_msg:
|
|
461
469
|
break
|
|
@@ -474,7 +482,7 @@ def main():
|
|
|
474
482
|
if hasattr(sys.stdout, 'reconfigure'):
|
|
475
483
|
sys.stdout.reconfigure(encoding='utf-8')
|
|
476
484
|
|
|
477
|
-
parser = argparse.ArgumentParser(description="
|
|
485
|
+
parser = argparse.ArgumentParser(description="LLM Proxy CLI: A fault-tolerant CLI tool for LLM delegation.")
|
|
478
486
|
parser.add_argument("-v", "--version", action="version", version=f"LLM Proxy CLI v{__version__}")
|
|
479
487
|
parser.add_argument("-m", "--models", required=True, help="Comma-separated list of provider:model fallbacks (e.g. nvidia:nemotron,groq:llama3, or groq:auto-smart / groq:auto-fast).")
|
|
480
488
|
parser.add_argument("-p", "--prompt", help="The prompt text to send to the model.")
|
|
@@ -506,9 +514,8 @@ def main():
|
|
|
506
514
|
|
|
507
515
|
cb = CircuitBreaker(args.project, args.max_failures, args.cooldown)
|
|
508
516
|
try:
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
print(response)
|
|
517
|
+
query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
|
|
518
|
+
print()
|
|
512
519
|
except Exception as e:
|
|
513
520
|
logger.error(str(e))
|
|
514
521
|
sys.exit(1)
|
|
@@ -46,3 +46,46 @@ def test_circuit_breaker_success_reset(tmp_path):
|
|
|
46
46
|
cb.record_success("modelC")
|
|
47
47
|
circuit = cb.load()
|
|
48
48
|
assert "modelC" not in circuit or circuit["modelC"]["failures"] == 0
|
|
49
|
+
|
|
50
|
+
import time
|
|
51
|
+
|
|
52
|
+
def test_circuit_breaker_decay():
|
|
53
|
+
cb = CircuitBreaker("decay_test", max_failures=2, cooldown_seconds=60)
|
|
54
|
+
|
|
55
|
+
# Clean state
|
|
56
|
+
if os.path.exists(cb.circuit_file):
|
|
57
|
+
os.remove(cb.circuit_file)
|
|
58
|
+
|
|
59
|
+
import llm_proxy_cli
|
|
60
|
+
|
|
61
|
+
# Mock time.time to simulate passage of time
|
|
62
|
+
original_time = time.time
|
|
63
|
+
current_mock_time = original_time()
|
|
64
|
+
|
|
65
|
+
def mock_time():
|
|
66
|
+
return current_mock_time
|
|
67
|
+
|
|
68
|
+
llm_proxy_cli.time.time = mock_time
|
|
69
|
+
|
|
70
|
+
try:
|
|
71
|
+
# Failure 1 at T=0
|
|
72
|
+
cb.record_failure("test:model")
|
|
73
|
+
assert cb.check_health("test:model") == True
|
|
74
|
+
|
|
75
|
+
# Advance time by 400 seconds (past 300s decay threshold)
|
|
76
|
+
current_mock_time += 400
|
|
77
|
+
|
|
78
|
+
# Failure 2 at T=400
|
|
79
|
+
# This should reset the counter to 0 before adding 1, so it won't trip (max_failures=2)
|
|
80
|
+
cb.record_failure("test:model")
|
|
81
|
+
assert cb.check_health("test:model") == True
|
|
82
|
+
|
|
83
|
+
# Failure 3 immediately after (T=400)
|
|
84
|
+
# Counter becomes 2 -> trips!
|
|
85
|
+
cb.record_failure("test:model")
|
|
86
|
+
assert cb.check_health("test:model") == False
|
|
87
|
+
|
|
88
|
+
finally:
|
|
89
|
+
llm_proxy_cli.time.time = original_time
|
|
90
|
+
if os.path.exists(cb.circuit_file):
|
|
91
|
+
os.remove(cb.circuit_file)
|
|
@@ -197,3 +197,38 @@ def test_query_ai_resolves_auto_alias_before_calling_provider(monkeypatch, mock_
|
|
|
197
197
|
# circuit breaker must key on the resolved model, never on the literal "auto-smart" alias
|
|
198
198
|
mock_cb.check_health.assert_called_with("groq:resolved-model-xyz")
|
|
199
199
|
mock_cb.record_success.assert_called_once_with("groq:resolved-model-xyz")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def test_query_ai_streaming_output(capsys, monkeypatch, mock_cb):
|
|
203
|
+
"""Test that stream_out=True sends reasoning to stderr and content to stdout, exactly once."""
|
|
204
|
+
monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": "dummy", "api_key": "key"})
|
|
205
|
+
|
|
206
|
+
mock_client = MagicMock()
|
|
207
|
+
mock_chunk1 = MagicMock()
|
|
208
|
+
mock_chunk1.choices = [MagicMock()]
|
|
209
|
+
mock_chunk1.choices[0].delta.reasoning_content = "thinking..."
|
|
210
|
+
mock_chunk1.choices[0].delta.content = ""
|
|
211
|
+
|
|
212
|
+
mock_chunk2 = MagicMock()
|
|
213
|
+
mock_chunk2.choices = [MagicMock()]
|
|
214
|
+
mock_chunk2.choices[0].delta.reasoning_content = ""
|
|
215
|
+
mock_chunk2.choices[0].delta.content = "hello world"
|
|
216
|
+
|
|
217
|
+
mock_client.chat.completions.create.return_value = [mock_chunk1, mock_chunk2]
|
|
218
|
+
|
|
219
|
+
factory = openai_factory({"dummy": mock_client})
|
|
220
|
+
with patch.object(router, "OpenAI", side_effect=factory):
|
|
221
|
+
response = router.query_ai("groq:llama", "hi", mock_cb, stream_out=True)
|
|
222
|
+
|
|
223
|
+
captured = capsys.readouterr()
|
|
224
|
+
|
|
225
|
+
# 1. Stdout must only contain "hello world" (no reasoning)
|
|
226
|
+
assert "hello world" in captured.out
|
|
227
|
+
assert "thinking..." not in captured.out
|
|
228
|
+
|
|
229
|
+
# 2. Stderr must contain the reasoning
|
|
230
|
+
assert "thinking..." in captured.err
|
|
231
|
+
|
|
232
|
+
# 3. The returned string still contains everything for library usage
|
|
233
|
+
assert "hello world" in response
|
|
234
|
+
assert "thinking..." in response
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|