llm-proxy-cli 0.5.1__tar.gz → 0.5.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-proxy-cli
3
- Version: 0.5.1
3
+ Version: 0.5.3
4
4
  Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
5
5
  Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
6
6
  Classifier: Programming Language :: Python :: 3
@@ -29,7 +29,7 @@ Dynamic: license-file
29
29
  - **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
30
30
  - **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
31
31
  - **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
32
- - **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
32
+ - **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
33
33
  - **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
34
34
 
35
35
  ## 🏗️ Architecture & Under the Hood
@@ -13,7 +13,7 @@
13
13
  - **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
14
14
  - **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
15
15
  - **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
16
- - **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
16
+ - **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
17
17
  - **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
18
18
 
19
19
  ## 🏗️ Architecture & Under the Hood
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: llm-proxy-cli
3
- Version: 0.5.1
3
+ Version: 0.5.3
4
4
  Summary: A lightweight CLI tool for delegating LLM tasks to expert models across multiple providers.
5
5
  Author-email: Kerem Barbaros Karnabat <kbarbaros@hotmail.com>
6
6
  Classifier: Programming Language :: Python :: 3
@@ -29,7 +29,7 @@ Dynamic: license-file
29
29
  - **Active Liveness Verification**: Pings candidates with a minimal chat request to drop fake/gated models before they crash your task.
30
30
  - **Automatic Fallbacks**: Provide a comma-separated list of models. If one fails, it instantly falls back to the next.
31
31
  - **Circuit Breaker**: Built-in health tracking and cooldowns to prevent spamming dead endpoints.
32
- - **Reasoning Extraction**: Automatically extracts and formats hidden `<thought>` or `reasoning` blocks (e.g., from Nemotron).
32
+ - **Reasoning Extraction**: Automatically extracts `reasoning_content` from natively supported models (e.g., DeepSeek-R1 or Nemotron) and outputs them to stderr.
33
33
  - **Streaming Native**: Built on the official OpenAI SDK for fast and reliable streaming chunks.
34
34
 
35
35
  ## 🏗️ Architecture & Under the Hood
@@ -9,7 +9,7 @@ import logging
9
9
  from openai import OpenAI
10
10
  from filelock import FileLock, Timeout
11
11
 
12
- __version__ = "0.5.1"
12
+ __version__ = "0.5.3"
13
13
 
14
14
  # Optional import for anthropic
15
15
  try:
@@ -43,14 +43,15 @@ AUTO_DISCOVERY_TIMEOUT = 5 # seconds - keep the "auto" resolve snappy
43
43
 
44
44
  def get_api_key(provider):
45
45
  keys_file = os.path.join(get_config_dir(), "keys.json")
46
+ keys = {}
46
47
  if os.path.exists(keys_file):
47
48
  try:
48
49
  with open(keys_file, 'r', encoding='utf-8') as f:
49
50
  keys = json.load(f)
50
- env_name = f"{provider.upper()}_API_KEY"
51
51
  except Exception:
52
52
  pass
53
- return os.environ.get(f"{provider.upper()}_API_KEY") or (keys.get(env_name) if 'keys' in locals() else None)
53
+ env_name = f"{provider.upper()}_API_KEY"
54
+ return os.environ.get(env_name) or keys.get(env_name)
54
55
 
55
56
 
56
57
  PROVIDERS = {
@@ -336,8 +337,16 @@ class CircuitBreaker:
336
337
  with FileLock(self.lock_file, timeout=5):
337
338
  circuit = self.load()
338
339
  if model_id not in circuit:
339
- circuit[model_id] = {'failures': 0, 'cooldown_until': 0}
340
+ circuit[model_id] = {'failures': 0, 'cooldown_until': 0, 'last_failure_time': 0}
341
+
342
+ # Decay old failures after 300 seconds
343
+ last_fail = circuit[model_id].get('last_failure_time', 0)
344
+ if time.time() - last_fail > 300:
345
+ circuit[model_id]['failures'] = 0
346
+
340
347
  circuit[model_id]['failures'] += 1
348
+ circuit[model_id]['last_failure_time'] = time.time()
349
+
341
350
  if circuit[model_id]['failures'] >= self.max_failures:
342
351
  circuit[model_id]['cooldown_until'] = time.time() + self.cooldown_seconds
343
352
  circuit[model_id]['failures'] = 0
@@ -417,7 +426,6 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
417
426
  ) as stream:
418
427
  for text in stream.text_stream:
419
428
  if stream_out:
420
- import sys
421
429
  sys.stdout.write(text)
422
430
  sys.stdout.flush()
423
431
  full_content += text
@@ -436,13 +444,11 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
436
444
  if reasoning:
437
445
  full_reasoning += reasoning
438
446
  if stream_out:
439
- import sys
440
447
  sys.stderr.write(reasoning)
441
448
  sys.stderr.flush()
442
449
  content = chunk.choices[0].delta.content
443
450
  if content:
444
451
  if stream_out:
445
- import sys
446
452
  sys.stdout.write(content)
447
453
  sys.stdout.flush()
448
454
  full_content += content
@@ -456,6 +462,9 @@ def query_ai(models_list, prompt, cb: CircuitBreaker, max_retries=2, base_timeou
456
462
  except Exception as e:
457
463
  error_msg = str(e).lower()
458
464
  logger.error(f"Attempt {attempt+1} failed for {resolved_key}: {str(e)}")
465
+ if stream_out and (full_content or full_reasoning):
466
+ sys.stderr.write(f"\n[STREAM INTERRUPTED: {str(e)} - FALLBACK TRIGGERED]\n")
467
+ sys.stderr.flush()
459
468
  status_code = getattr(e, "status_code", None)
460
469
  if status_code in (404, 401, 403) or "404" in error_msg or "not found" in error_msg or "auth" in error_msg:
461
470
  break
@@ -474,7 +483,7 @@ def main():
474
483
  if hasattr(sys.stdout, 'reconfigure'):
475
484
  sys.stdout.reconfigure(encoding='utf-8')
476
485
 
477
- parser = argparse.ArgumentParser(description="Smart Router: A fault-tolerant CLI tool for LLM delegation.")
486
+ parser = argparse.ArgumentParser(description="LLM Proxy CLI: A fault-tolerant CLI tool for LLM delegation.")
478
487
  parser.add_argument("-v", "--version", action="version", version=f"LLM Proxy CLI v{__version__}")
479
488
  parser.add_argument("-m", "--models", required=True, help="Comma-separated list of provider:model fallbacks (e.g. nvidia:nemotron,groq:llama3, or groq:auto-smart / groq:auto-fast).")
480
489
  parser.add_argument("-p", "--prompt", help="The prompt text to send to the model.")
@@ -506,9 +515,8 @@ def main():
506
515
 
507
516
  cb = CircuitBreaker(args.project, args.max_failures, args.cooldown)
508
517
  try:
509
- response = query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
510
- if not sys.stdout.isatty():
511
- print(response)
518
+ query_ai(args.models, prompt_text, cb, force_refresh_auto=args.refresh_models, stream_out=True)
519
+ print()
512
520
  except Exception as e:
513
521
  logger.error(str(e))
514
522
  sys.exit(1)
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "llm-proxy-cli"
7
- version = "0.5.1"
7
+ version = "0.5.3"
8
8
  authors = [
9
9
  { name="Kerem Barbaros Karnabat", email="kbarbaros@hotmail.com" }
10
10
  ]
@@ -46,3 +46,46 @@ def test_circuit_breaker_success_reset(tmp_path):
46
46
  cb.record_success("modelC")
47
47
  circuit = cb.load()
48
48
  assert "modelC" not in circuit or circuit["modelC"]["failures"] == 0
49
+
50
+ import time
51
+
52
+ def test_circuit_breaker_decay():
53
+ cb = CircuitBreaker("decay_test", max_failures=2, cooldown_seconds=60)
54
+
55
+ # Clean state
56
+ if os.path.exists(cb.circuit_file):
57
+ os.remove(cb.circuit_file)
58
+
59
+ import llm_proxy_cli
60
+
61
+ # Mock time.time to simulate passage of time
62
+ original_time = time.time
63
+ current_mock_time = original_time()
64
+
65
+ def mock_time():
66
+ return current_mock_time
67
+
68
+ llm_proxy_cli.time.time = mock_time
69
+
70
+ try:
71
+ # Failure 1 at T=0
72
+ cb.record_failure("test:model")
73
+ assert cb.check_health("test:model") == True
74
+
75
+ # Advance time by 400 seconds (past 300s decay threshold)
76
+ current_mock_time += 400
77
+
78
+ # Failure 2 at T=400
79
+ # This should reset the counter to 0 before adding 1, so it won't trip (max_failures=2)
80
+ cb.record_failure("test:model")
81
+ assert cb.check_health("test:model") == True
82
+
83
+ # Failure 3 immediately after (T=400)
84
+ # Counter becomes 2 -> trips!
85
+ cb.record_failure("test:model")
86
+ assert cb.check_health("test:model") == False
87
+
88
+ finally:
89
+ llm_proxy_cli.time.time = original_time
90
+ if os.path.exists(cb.circuit_file):
91
+ os.remove(cb.circuit_file)
@@ -197,3 +197,38 @@ def test_query_ai_resolves_auto_alias_before_calling_provider(monkeypatch, mock_
197
197
  # circuit breaker must key on the resolved model, never on the literal "auto-smart" alias
198
198
  mock_cb.check_health.assert_called_with("groq:resolved-model-xyz")
199
199
  mock_cb.record_success.assert_called_once_with("groq:resolved-model-xyz")
200
+
201
+
202
+ def test_query_ai_streaming_output(capsys, monkeypatch, mock_cb):
203
+ """Test that stream_out=True sends reasoning to stderr and content to stdout, exactly once."""
204
+ monkeypatch.setitem(router.PROVIDERS, "groq", {"base_url": "dummy", "api_key": "key"})
205
+
206
+ mock_client = MagicMock()
207
+ mock_chunk1 = MagicMock()
208
+ mock_chunk1.choices = [MagicMock()]
209
+ mock_chunk1.choices[0].delta.reasoning_content = "thinking..."
210
+ mock_chunk1.choices[0].delta.content = ""
211
+
212
+ mock_chunk2 = MagicMock()
213
+ mock_chunk2.choices = [MagicMock()]
214
+ mock_chunk2.choices[0].delta.reasoning_content = ""
215
+ mock_chunk2.choices[0].delta.content = "hello world"
216
+
217
+ mock_client.chat.completions.create.return_value = [mock_chunk1, mock_chunk2]
218
+
219
+ factory = openai_factory({"dummy": mock_client})
220
+ with patch.object(router, "OpenAI", side_effect=factory):
221
+ response = router.query_ai("groq:llama", "hi", mock_cb, stream_out=True)
222
+
223
+ captured = capsys.readouterr()
224
+
225
+ # 1. Stdout must only contain "hello world" (no reasoning)
226
+ assert "hello world" in captured.out
227
+ assert "thinking..." not in captured.out
228
+
229
+ # 2. Stderr must contain the reasoning
230
+ assert "thinking..." in captured.err
231
+
232
+ # 3. The returned string still contains everything for library usage
233
+ assert "hello world" in response
234
+ assert "thinking..." in response
File without changes
File without changes