opsiom-cli 1.2.4__tar.gz → 1.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/PKG-INFO +1 -1
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/pyproject.toml +1 -1
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli/cli.py +148 -33
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/PKG-INFO +1 -1
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/README.md +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/setup.cfg +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli/__init__.py +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/SOURCES.txt +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/dependency_links.txt +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/entry_points.txt +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/requires.txt +0 -0
- {opsiom_cli-1.2.4 → opsiom_cli-1.2.5}/src/opsiom_cli.egg-info/top_level.txt +0 -0
|
@@ -23,7 +23,6 @@ import json
|
|
|
23
23
|
import time
|
|
24
24
|
import shutil
|
|
25
25
|
import hashlib
|
|
26
|
-
import textwrap
|
|
27
26
|
import argparse
|
|
28
27
|
import threading
|
|
29
28
|
import itertools
|
|
@@ -40,7 +39,7 @@ try:
|
|
|
40
39
|
except ImportError:
|
|
41
40
|
readline = None # absent par défaut sur certains Windows (pyreadline3 comble le manque)
|
|
42
41
|
|
|
43
|
-
VERSION = "1.2.
|
|
42
|
+
VERSION = "1.2.5"
|
|
44
43
|
CONFIG_PATH = Path.home() / ".config" / "opsiom" / "config.json"
|
|
45
44
|
HISTORY_PATH = Path.home() / ".local" / "share" / "opsiom" / "history"
|
|
46
45
|
TRANSCRIPTS_DIR = Path.home() / "opsiom-conversations"
|
|
@@ -281,11 +280,13 @@ def resolve_settings(args):
|
|
|
281
280
|
|
|
282
281
|
|
|
283
282
|
# ----------------------------------------------------------------------------
|
|
284
|
-
# Quota
|
|
285
|
-
#
|
|
286
|
-
#
|
|
287
|
-
#
|
|
288
|
-
#
|
|
283
|
+
# Quota — 500 tokens/jour par clé API. Le fichier usage.json ne sert plus
|
|
284
|
+
# qu'à AFFICHER un dernier quota connu avant la toute première requête de la
|
|
285
|
+
# session (utile pour /quota juste après le lancement) : dès qu'une réponse
|
|
286
|
+
# arrive, c'est le "remaining_quota" renvoyé par le serveur (lui-même décompté
|
|
287
|
+
# côté Octix) qui fait foi et remplace la valeur locale. Un utilisateur qui
|
|
288
|
+
# modifierait usage.json à la main ne gagnerait donc plus rien : le serveur
|
|
289
|
+
# refuse désormais les requêtes une fois son propre compteur à zéro (429).
|
|
289
290
|
# ----------------------------------------------------------------------------
|
|
290
291
|
|
|
291
292
|
def _api_key_id(api_key):
|
|
@@ -333,14 +334,18 @@ def get_quota_state(api_key):
|
|
|
333
334
|
return used, max(0, DAILY_TOKEN_QUOTA - used)
|
|
334
335
|
|
|
335
336
|
|
|
336
|
-
def
|
|
337
|
+
def record_authoritative_remaining(api_key, remaining):
|
|
338
|
+
"""Remplace le compteur local par le quota restant renvoyé par le
|
|
339
|
+
serveur (lui-même décompté côté Octix) — c'est la seule valeur qui
|
|
340
|
+
compte désormais, le fichier local n'est qu'un cache d'affichage."""
|
|
341
|
+
if remaining is None:
|
|
342
|
+
return
|
|
337
343
|
usage = load_usage()
|
|
338
344
|
key_id = _api_key_id(api_key)
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
usage[key_id] = entry
|
|
345
|
+
usage[key_id] = {
|
|
346
|
+
"date": today_str(),
|
|
347
|
+
"tokens_used": max(0, DAILY_TOKEN_QUOTA - remaining),
|
|
348
|
+
}
|
|
344
349
|
save_usage(usage)
|
|
345
350
|
|
|
346
351
|
|
|
@@ -377,6 +382,33 @@ class OpsiomClient:
|
|
|
377
382
|
r.raise_for_status()
|
|
378
383
|
return r.json()
|
|
379
384
|
|
|
385
|
+
def chat_stream(self, message, model=None, **params):
|
|
386
|
+
"""Générateur : cède chaque fragment de texte au fur et à mesure,
|
|
387
|
+
puis le dict final {'done', 'model', 'tokens_used', 'remaining_quota'}.
|
|
388
|
+
Lève requests.HTTPError si le serveur refuse la requête (401/429/...) —
|
|
389
|
+
dans ce cas rien n'a encore été cédé, l'appelant peut afficher l'erreur
|
|
390
|
+
comme pour un appel non-streamé.
|
|
391
|
+
"""
|
|
392
|
+
payload = {"message": message}
|
|
393
|
+
if model:
|
|
394
|
+
payload["model"] = model
|
|
395
|
+
payload.update(params)
|
|
396
|
+
r = self.session.post(
|
|
397
|
+
f"{self.base_url}/api/chat/stream", json=payload,
|
|
398
|
+
timeout=self.timeout, stream=True,
|
|
399
|
+
)
|
|
400
|
+
r.raise_for_status()
|
|
401
|
+
for line in r.iter_lines(decode_unicode=True):
|
|
402
|
+
if not line or not line.startswith("data: "):
|
|
403
|
+
continue
|
|
404
|
+
event = json.loads(line[len("data: "):])
|
|
405
|
+
if event.get("done"):
|
|
406
|
+
yield event
|
|
407
|
+
return
|
|
408
|
+
token = event.get("token")
|
|
409
|
+
if token:
|
|
410
|
+
yield token
|
|
411
|
+
|
|
380
412
|
|
|
381
413
|
# ----------------------------------------------------------------------------
|
|
382
414
|
# Interface interactive
|
|
@@ -454,13 +486,59 @@ def select_model_interactively(models_payload, current_model):
|
|
|
454
486
|
return resolved
|
|
455
487
|
|
|
456
488
|
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
489
|
+
class StreamPrinter:
|
|
490
|
+
"""Affiche la réponse au fur et à mesure qu'elle arrive (streaming),
|
|
491
|
+
avec un retour à la ligne automatique à la largeur du terminal.
|
|
492
|
+
|
|
493
|
+
Contrairement à un `textwrap.fill` relancé à chaque fragment reçu (qui
|
|
494
|
+
réimprimerait tout depuis le début et ferait scintiller le terminal),
|
|
495
|
+
on ne pousse en sortie que du texte jamais encore affiché, mot par mot.
|
|
496
|
+
"""
|
|
497
|
+
|
|
498
|
+
def __init__(self, indent=" "):
|
|
499
|
+
self.indent = indent
|
|
500
|
+
self.width = max(term_width() - len(indent), 30)
|
|
501
|
+
self.line_len = 0
|
|
502
|
+
self.buffer = ""
|
|
503
|
+
self.started = False
|
|
504
|
+
|
|
505
|
+
def _flush_word(self, word):
|
|
506
|
+
if not self.started:
|
|
507
|
+
sys.stdout.write("\n" + self.indent)
|
|
508
|
+
self.started = True
|
|
509
|
+
if self.line_len > 0 and self.line_len + 1 + len(word) > self.width:
|
|
510
|
+
sys.stdout.write("\n" + self.indent)
|
|
511
|
+
self.line_len = 0
|
|
512
|
+
elif self.line_len > 0:
|
|
513
|
+
sys.stdout.write(" ")
|
|
514
|
+
self.line_len += 1
|
|
515
|
+
sys.stdout.write(word)
|
|
516
|
+
self.line_len += len(word)
|
|
517
|
+
sys.stdout.flush()
|
|
518
|
+
|
|
519
|
+
def feed(self, chunk):
|
|
520
|
+
self.buffer += chunk
|
|
521
|
+
while True:
|
|
522
|
+
idx = next((i for i, c in enumerate(self.buffer) if c in " \n"), None)
|
|
523
|
+
if idx is None:
|
|
524
|
+
break
|
|
525
|
+
word, sep = self.buffer[:idx], self.buffer[idx]
|
|
526
|
+
if word:
|
|
527
|
+
self._flush_word(word)
|
|
528
|
+
if sep == "\n":
|
|
529
|
+
if not self.started:
|
|
530
|
+
sys.stdout.write("\n" + self.indent)
|
|
531
|
+
self.started = True
|
|
532
|
+
sys.stdout.write("\n" + self.indent)
|
|
533
|
+
self.line_len = 0
|
|
534
|
+
sys.stdout.flush()
|
|
535
|
+
self.buffer = self.buffer[idx + 1:]
|
|
536
|
+
|
|
537
|
+
def close(self):
|
|
538
|
+
if self.buffer:
|
|
539
|
+
self._flush_word(self.buffer)
|
|
540
|
+
self.buffer = ""
|
|
541
|
+
print()
|
|
464
542
|
|
|
465
543
|
|
|
466
544
|
def run_chat(client: OpsiomClient, url, api_key):
|
|
@@ -559,37 +637,74 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
559
637
|
spinner = Spinner(label="Réflexion")
|
|
560
638
|
spinner.start()
|
|
561
639
|
start_time = time.perf_counter()
|
|
640
|
+
printer = StreamPrinter()
|
|
641
|
+
response_parts = []
|
|
642
|
+
final_event = {}
|
|
643
|
+
spinner_running = True
|
|
562
644
|
try:
|
|
563
|
-
|
|
645
|
+
for piece in client.chat_stream(user_input, model=current_model):
|
|
646
|
+
if isinstance(piece, dict):
|
|
647
|
+
final_event = piece
|
|
648
|
+
break
|
|
649
|
+
if spinner_running:
|
|
650
|
+
spinner.stop()
|
|
651
|
+
spinner_running = False
|
|
652
|
+
response_parts.append(piece)
|
|
653
|
+
printer.feed(piece)
|
|
564
654
|
except KeyboardInterrupt:
|
|
565
|
-
|
|
566
|
-
|
|
655
|
+
if spinner_running:
|
|
656
|
+
spinner.stop()
|
|
657
|
+
print(muted("\n (requête annulée) "))
|
|
567
658
|
continue
|
|
568
659
|
except requests.HTTPError as e:
|
|
569
|
-
|
|
660
|
+
if spinner_running:
|
|
661
|
+
spinner.stop()
|
|
570
662
|
if e.response is not None and e.response.status_code in (401, 403):
|
|
571
663
|
print_error("clé API invalide ou expirée.", "reconfigure avec : opsiom --configure")
|
|
572
664
|
elif e.response is not None and e.response.status_code == 429:
|
|
573
|
-
|
|
665
|
+
detail = None
|
|
666
|
+
try:
|
|
667
|
+
detail = e.response.json().get("error")
|
|
668
|
+
except (ValueError, AttributeError):
|
|
669
|
+
pass
|
|
670
|
+
print_error("quota dépassé côté serveur.", detail or "réessaie plus tard")
|
|
574
671
|
else:
|
|
575
672
|
print_error("la requête a échoué ", str(e))
|
|
576
673
|
print()
|
|
577
674
|
continue
|
|
578
675
|
except requests.RequestException as e:
|
|
579
|
-
|
|
676
|
+
if spinner_running:
|
|
677
|
+
spinner.stop()
|
|
580
678
|
print_error("la requête a échoué ", str(e))
|
|
581
679
|
print()
|
|
582
680
|
continue
|
|
583
|
-
|
|
681
|
+
|
|
682
|
+
if spinner_running:
|
|
683
|
+
spinner.stop() # réponse vide reçue sans le moindre fragment
|
|
684
|
+
printer.close()
|
|
584
685
|
elapsed_ms = int((time.perf_counter() - start_time) * 1000)
|
|
585
686
|
|
|
586
|
-
response_text =
|
|
587
|
-
used_model =
|
|
588
|
-
|
|
687
|
+
response_text = "".join(response_parts) or "(réponse vide)"
|
|
688
|
+
used_model = final_event.get("model", current_model or "?")
|
|
689
|
+
print(f" {muted(format_elapsed(elapsed_ms))}")
|
|
589
690
|
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
691
|
+
remaining_quota = final_event.get("remaining_quota")
|
|
692
|
+
if remaining_quota is not None:
|
|
693
|
+
record_authoritative_remaining(api_key, remaining_quota)
|
|
694
|
+
else:
|
|
695
|
+
# Le serveur n'a pas renvoyé de quota faisant autorité (Octix
|
|
696
|
+
# injoignable côté serveur, p. ex.) : on retombe sur l'estimation
|
|
697
|
+
# locale plutôt que de ne rien afficher.
|
|
698
|
+
estimated = estimate_tokens(user_input) + estimate_tokens(response_text)
|
|
699
|
+
usage = load_usage()
|
|
700
|
+
key_id = _api_key_id(api_key)
|
|
701
|
+
entry = usage.get(key_id, {})
|
|
702
|
+
if entry.get("date") != today_str():
|
|
703
|
+
entry = {"date": today_str(), "tokens_used": 0}
|
|
704
|
+
entry["tokens_used"] = entry.get("tokens_used", 0) + estimated
|
|
705
|
+
usage[key_id] = entry
|
|
706
|
+
save_usage(usage)
|
|
707
|
+
_, remaining_quota = get_quota_state(api_key)
|
|
593
708
|
print(f" {muted(f'{remaining_quota} tokens restants aujourd’hui')}")
|
|
594
709
|
print()
|
|
595
710
|
transcript.append({"role": "assistant", "text": response_text, "model": used_model, "elapsed_ms": elapsed_ms})
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|