llm-proxy-cli 0.5.1__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-proxy-cli
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
5
5
  Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
6
6
  Classifier: Programming Language :: Python :: 3
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-proxy-cli
3
- Version: 0.5.1
3
+ Version: 0.5.2
4
4
  Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
5
5
  Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
6
6
  Classifier: Programming Language :: Python :: 3
@@ -9,7 +9,7 @@ import logging
9
9
  from openai import OpenAI
10
10
  from filelock import FileLock, Timeout
11
11
 
12
- __version__ = "0.5.1"
12
+ __version__ = "0.5.2"
13
13
 
14
14
  # Optional import for anthropic
15
15
  try:
@@ -336,8 +336,16 @@ class CircuitBreaker:
336
336
  with FileLock(self.lock_file, timeout=5):
337
337
  circuit = self.load()
338
338
  if model_id not in circuit:
339
- circuit[model_id] = {'failures': 0, 'cooldown_until': 0}
339
+ circuit[model_id] = {'failures': 0, 'cooldown_until': 0, 'last_failure_time': 0}
340
+
341
+ # Decay old failures after 300 seconds
342
+ last_fail = circuit[model_id].get('last_failure_time', 0)
343
+ if time.time() - last_fail > 300:
344
+ circuit[model_id]['failures'] = 0
345
+
340
346
  circuit[model_id]['failures'] += 1
347
+ circuit[model_id]['last_failure_time'] = time.time()
348
+
341
349
  if circuit[model_id]['failures'] >= self.max_failures:
342
350
  circuit[model_id]['cooldown_until'] = time.time() + self.cooldown_seconds
343
351
  circuit[model_id]['failures'] = 0
@@ -417,7 +425,6 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
417
425
  ) as stream:
418
426
  for text in stream.text_stream:
419
427
  if stream_out:
420
- import sys
421
428
  sys.stdout.write(text)
422
429
  sys.stdout.flush()
423
430
  full_content += text
@@ -436,13 +443,11 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
436
443
  if reasoning:
437
444
  full_reasoning += reasoning
438
445
  if stream_out:
439
- import sys
440
446
  sys.stderr.write(reasoning)
441
447
  sys.stderr.flush()
442
448
  content = chunk.choices[0].delta.content
443
449
  if content:
444
450
  if stream_out:
445
- import sys
446
451
  sys.stdout.write(content)
447
452
  sys.stdout.flush()
448
453
  full_content += content
@@ -456,6 +461,9 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
456
461
  except Exception as e:
457
462
  error_msg = str(e).lower()
458
463
  logger.error(f"Attempt {attempt+1} failed for {resolved_key}: {str(e)}")
464
+ if stream_out and (full_content or full_reasoning):
465
+ sys.stderr.write(f"\n[STREAM INTERRUPTED: {str(e)} - FALLBACK TRIGGERED]\n")
466
+ sys.stderr.flush()
459
467
  status_code = getattr(e, "status_code", None)
460
468
  if status_code in (404, 401, 403) or "404" in error_msg or "not found" in error_msg or "auth" in error_msg:
461
469
  break
@@ -474,7 +482,7 @@ def main():
474
482
  if hasattr(sys.stdout, 'reconfigure'):
475
483
  sys.stdout.reconfigure(encoding='utf-8')
476
484
 
477
- parser = argparse.ArgumentParser(description="Smart Router: A fault-tolerant CLI tool for LLM delegation.")
485
+ parser = argparse.ArgumentParser(description="LLM Proxy CLI: A fault-tolerant CLI tool for LLM delegation.")
478
486
  parser.add_argument("-v", "--version", action="version", version=f"LLM Proxy CLI v{__version__}")
479
487
  parser.add_argument("-m", "--models", required=True, help="Comma-separated list of provider:model fallbacks (e.g. nvidia:nemotron,groq:llama3, or groq:auto-smart / groq:auto-fast).")
480
488
  parser.add_argument("-p", "--prompt", help="The prompt text to send to the model.")
@@ -506,9 +514,8 @@ def main():
506
514
 
507
515
  cb = CircuitBreaker(args.project, args.max_failures, args.cooldown)
508
516
  try:
509
- response = query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
510
- if not sys.stdout.isatty():
511
- print(response)
517
+ query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
518
+ print()
512
519
  except Exception as e:
513
520
  logger.error(str(e))
514
521
  sys.exit(1)
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "llm-proxy-cli"
7
- version = "0.5.1"
7
+ version = "0.5.2"
8
8
  authors = [
9
9
  { name="Kerem Barbaros Karnabat", email="kbarbaros@hotmail.com" }
10
10
  ]
@@ -46,3 +46,46 @@ def test_circuit_breaker_success_reset(tmp_path):
46
46
  cb.record_success("modelC")
47
47
  circuit = cb.load()
48
48
  assert "modelC" not in circuit or circuit["modelC"]["failures"] == 0
49
+
50
+ import time
51
+
52
+ def test_circuit_breaker_decay():
53
+ cb = CircuitBreaker("decay_test", max_failures=2, cooldown_seconds=60)
54
+
55
+ # Clean state
56
+ if os.path.exists(cb.circuit_file):
57
+ os.remove(cb.circuit_file)
58
+
59
+ import llm_proxy_cli
60
+
61
+ # Mock time.time to simulate passage of time
62
+ original_time = time.time
63
+ current_mock_time = original_time()
64
+
65
+ def mock_time():
66
+ return current_mock_time
67
+
68
+ llm_proxy_cli.time.time = mock_time
69
+
70
+ try:
71
+ # Failure 1 at T=0
72
+ cb.record_failure("test:model")
73
+ assert cb.check_health("test:model") == True
74
+
75
+ # Advance time by 400 seconds (past 300s decay threshold)
76
+ current_mock_time += 400
77
+
78
+ # Failure 2 at T=400
79
+ # This should reset the counter to 0 before adding 1, so it won't trip (max_failures=2)
80
+ cb.record_failure("test:model")
81
+ assert cb.check_health("test:model") == True
82
+
83
+ # Failure 3 immediately after (T=400)
84
+ # Counter becomes 2 -> trips!
85
+ cb.record_failure("test:model")
86
+ assert cb.check_health("test:model") == False
87
+
88
+ finally:
89
+ llm_proxy_cli.time.time = original_time
90
+ if os.path.exists(cb.circuit_file):
91
+ os.remove(cb.circuit_file)
@@ -197,3 +197,38 @@ def test_query_ai_resolves_auto_alias_before_calling_provider(monkeypatch, mock_
197
197
  # circuit breaker must key on the resolved model, never on the literal "auto-smart" alias
198
198
  mock_cb.check_health.assert_called_with("groq:resolved-model-xyz")
199
199
  mock_cb.record_success.assert_called_once_with("groq:resolved-model-xyz")
200
+
201
+
202
+ def test_query_ai_streaming_output(capsys, monkeypatch, mock_cb):
203
+ """Test that stream_out=True sends reasoning to stderr and content to stdout, exactly once."""
204
+ monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": "dummy", "api_key": "key"})
205
+
206
+ mock_client = MagicMock()
207
+ mock_chunk1 = MagicMock()
208
+ mock_chunk1.choices = [MagicMock()]
209
+ mock_chunk1.choices[0].delta.reasoning_content = "thinking..."
210
+ mock_chunk1.choices[0].delta.content = ""
211
+
212
+ mock_chunk2 = MagicMock()
213
+ mock_chunk2.choices = [MagicMock()]
214
+ mock_chunk2.choices[0].delta.reasoning_content = ""
215
+ mock_chunk2.choices[0].delta.content = "hello world"
216
+
217
+ mock_client.chat.completions.create.return_value = [mock_chunk1, mock_chunk2]
218
+
219
+ factory = openai_factory({"dummy": mock_client})
220
+ with patch.object(router, "OpenAI", side_effect=factory):
221
+ response = router.query_ai("groq:llama", "hi", mock_cb, stream_out=True)
222
+
223
+ captured = capsys.readouterr()
224
+
225
+ # 1. Stdout must only contain "hello world" (no reasoning)
226
+ assert "hello world" in captured.out
227
+ assert "thinking..." not in captured.out
228
+
229
+ # 2. Stderr must contain the reasoning
230
+ assert "thinking..." in captured.err
231
+
232
+ # 3. The returned string still contains everything for library usage
233
+ assert "hello world" in response
234
+ assert "thinking..." in response
File without changes
File without changes
File without changes