opsiom-cli 1.2.5__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/PKG-INFO +1 -1
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/pyproject.toml +2 -2
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli/cli.py +391 -89
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/PKG-INFO +1 -1
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/README.md +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/setup.cfg +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli/__init__.py +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/SOURCES.txt +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/dependency_links.txt +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/entry_points.txt +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/requires.txt +0 -0
- {opsiom_cli-1.2.5 → opsiom_cli-1.4.0}/src/opsiom_cli.egg-info/top_level.txt +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "opsiom-cli"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.4.0"
|
|
8
8
|
description = "CLI de chat pour Opsiom, l'assistant IA francophone."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -20,4 +20,4 @@ Homepage = "https://github.com/ton-compte/opsiom-cli"
|
|
|
20
20
|
opsiom = "opsiom_cli.cli:main"
|
|
21
21
|
|
|
22
22
|
[tool.setuptools.packages.find]
|
|
23
|
-
where = ["src"]
|
|
23
|
+
where = ["src"]
|
|
@@ -10,23 +10,37 @@
|
|
|
10
10
|
# Dépendances : uniquement `requests`
|
|
11
11
|
# pip install requests
|
|
12
12
|
#
|
|
13
|
+
# Deux types de serveur sont gérés (détectés d'après l'URL, ou forcés avec
|
|
14
|
+
# --backend) :
|
|
15
|
+
# - "api" : le serveur d'inférence classique (ngrok, Render…), routes
|
|
16
|
+
# /api/*, authentification par clé API (en-tête X-API-Key) ;
|
|
17
|
+
# - "hf" : un Space Hugging Face (*.hf.space) qui fait tourner l'interface
|
|
18
|
+
# Opsiom, routes /status, /models, /chat/stream, authentification
|
|
19
|
+
# par compte Octix (pseudo + mot de passe, session par cookie).
|
|
20
|
+
#
|
|
13
21
|
# Configuration (aucune URL n'est codée en dur dans ce fichier) :
|
|
14
22
|
# Au premier lancement, le CLI demande l'URL du serveur (et une éventuelle
|
|
15
|
-
# clé API) puis les enregistre dans
|
|
16
|
-
#
|
|
17
|
-
#
|
|
23
|
+
# clé API, ou le pseudo Octix pour un Space) puis les enregistre dans
|
|
24
|
+
# ~/.config/opsiom/config.json. Le mot de passe n'est JAMAIS enregistré.
|
|
25
|
+
# Alternative : variables d'environnement OPSIOM_URL / OPSIOM_API_KEY /
|
|
26
|
+
# OPSIOM_USERNAME / OPSIOM_PASSWORD, ou options --url / --api-key /
|
|
27
|
+
# --username / --backend.
|
|
18
28
|
# ============================================================================
|
|
19
29
|
|
|
20
30
|
import os
|
|
31
|
+
import re
|
|
21
32
|
import sys
|
|
22
33
|
import json
|
|
23
34
|
import time
|
|
35
|
+
import html
|
|
24
36
|
import shutil
|
|
37
|
+
import getpass
|
|
25
38
|
import hashlib
|
|
26
39
|
import argparse
|
|
27
40
|
import threading
|
|
28
41
|
import itertools
|
|
29
42
|
from pathlib import Path
|
|
43
|
+
from urllib.parse import urlparse
|
|
30
44
|
|
|
31
45
|
try:
|
|
32
46
|
import requests
|
|
@@ -39,7 +53,7 @@ try:
|
|
|
39
53
|
except ImportError:
|
|
40
54
|
readline = None # absent par défaut sur certains Windows (pyreadline3 comble le manque)
|
|
41
55
|
|
|
42
|
-
VERSION = "1.
|
|
56
|
+
VERSION = "1.4.0"
|
|
43
57
|
CONFIG_PATH = Path.home() / ".config" / "opsiom" / "config.json"
|
|
44
58
|
HISTORY_PATH = Path.home() / ".local" / "share" / "opsiom" / "history"
|
|
45
59
|
TRANSCRIPTS_DIR = Path.home() / "opsiom-conversations"
|
|
@@ -128,7 +142,17 @@ OCTOPUS_MARK = [
|
|
|
128
142
|
|
|
129
143
|
|
|
130
144
|
|
|
131
|
-
def
|
|
145
|
+
def quota_exhausted(state):
|
|
146
|
+
return not state["unlimited"] and state["remaining"] <= 0
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def quota_summary(state):
|
|
150
|
+
if state["unlimited"]:
|
|
151
|
+
return "illimité"
|
|
152
|
+
return f"{state['used']}/{state['limit']} tokens aujourd’hui ({state['remaining']} restants)"
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def print_header(client, connected, models_loaded, current_model, show_mark=True):
|
|
132
156
|
print()
|
|
133
157
|
if show_mark:
|
|
134
158
|
for line in OCTOPUS_MARK:
|
|
@@ -138,12 +162,12 @@ def print_header(url, connected, api_key, models_loaded, current_model, show_mar
|
|
|
138
162
|
print(f" {muted('assistant IA francophone')}")
|
|
139
163
|
print()
|
|
140
164
|
status_symbol = ok_text("✓") if connected else error_text("✗")
|
|
141
|
-
print(f" {status_symbol} {muted('connecté à')} {
|
|
142
|
-
print(f" {muted('auth')} {muted(
|
|
143
|
-
if
|
|
144
|
-
|
|
145
|
-
quota_color = error_text if
|
|
146
|
-
print(f" {muted('quota')} {quota_color(
|
|
165
|
+
print(f" {status_symbol} {muted('connecté à')} {client.base_url}")
|
|
166
|
+
print(f" {muted('auth')} {muted(client.auth_label)}")
|
|
167
|
+
if client.tracks_quota:
|
|
168
|
+
state = get_quota_state(client.identity)
|
|
169
|
+
quota_color = error_text if quota_exhausted(state) else muted
|
|
170
|
+
print(f" {muted('quota')} {quota_color(quota_summary(state))}")
|
|
147
171
|
if models_loaded:
|
|
148
172
|
print(f" {muted('modèles')} {' · '.join(models_loaded)}")
|
|
149
173
|
if current_model:
|
|
@@ -255,38 +279,88 @@ def save_config(config):
|
|
|
255
279
|
CONFIG_PATH.write_text(json.dumps(config, indent=2))
|
|
256
280
|
|
|
257
281
|
|
|
258
|
-
def
|
|
282
|
+
def normalize_url(raw):
|
|
283
|
+
"""Nettoie l'URL saisie : ajoute le schéma manquant et transforme la page
|
|
284
|
+
d'un Space (https://huggingface.co/spaces/Auteur/Nom) en son URL directe
|
|
285
|
+
(https://auteur-nom.hf.space)."""
|
|
286
|
+
url = (raw or "").strip().strip("\"'").rstrip("/")
|
|
287
|
+
if not url:
|
|
288
|
+
return ""
|
|
289
|
+
if "://" not in url:
|
|
290
|
+
local = url.startswith(("localhost", "127.0.0.1"))
|
|
291
|
+
url = ("http://" if local else "https://") + url
|
|
292
|
+
parsed = urlparse(url)
|
|
293
|
+
match = re.match(r"^/spaces/([^/]+)/([^/]+)", parsed.path)
|
|
294
|
+
if parsed.netloc in ("huggingface.co", "www.huggingface.co") and match:
|
|
295
|
+
owner, name = (re.sub(r"[._]", "-", part).lower() for part in match.groups())
|
|
296
|
+
return f"https://{owner}-{name}.hf.space"
|
|
297
|
+
return url
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def detect_backend(url, preference="auto"):
|
|
301
|
+
"""'hf' pour un Space Hugging Face, 'api' sinon (ngrok, Render…).
|
|
302
|
+
Une préférence explicite ('api' ou 'hf') l'emporte sur la détection."""
|
|
303
|
+
if preference in ("api", "hf"):
|
|
304
|
+
return preference
|
|
305
|
+
host = (urlparse(url).hostname or "").lower()
|
|
306
|
+
return "hf" if host.endswith(".hf.space") or host == "huggingface.co" else "api"
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def prompt_for_config(backend_pref="auto"):
|
|
259
310
|
print(f" {accent('Configuration initiale', bold=True)}")
|
|
260
311
|
print(f" {muted('ces informations te sont transmises par la personne qui héberge Opsiom')} ")
|
|
261
|
-
url = input(" URL du serveur : ")
|
|
262
|
-
|
|
312
|
+
url = normalize_url(input(" URL du serveur (ngrok, ou Space Hugging Face) : "))
|
|
313
|
+
backend = detect_backend(url, backend_pref)
|
|
314
|
+
config = {"url": url, "backend": backend, "api_key": "", "username": ""}
|
|
315
|
+
if backend == "hf":
|
|
316
|
+
print(muted(" → Space Hugging Face : connexion avec ton compte Octix."))
|
|
317
|
+
config["username"] = input(" Pseudo Octix : ").strip()
|
|
318
|
+
else:
|
|
319
|
+
config["api_key"] = input(" Clé API (vide si aucune) : ").strip()
|
|
263
320
|
save = input(muted(" Sauvegarder localement pour la prochaine fois ? [O/n] ")).strip().lower()
|
|
264
|
-
config = {"url": url, "api_key": api_key}
|
|
265
321
|
if save in ("", "o", "oui", "y", "yes"):
|
|
266
322
|
save_config(config)
|
|
267
|
-
print(muted(f" → enregistré dans {CONFIG_PATH} "))
|
|
323
|
+
print(muted(f" → enregistré dans {CONFIG_PATH} (le mot de passe n'est jamais enregistré) "))
|
|
268
324
|
return config
|
|
269
325
|
|
|
270
326
|
|
|
271
327
|
def resolve_settings(args):
|
|
272
328
|
config = {} if args.configure else load_config()
|
|
273
|
-
|
|
329
|
+
url_override = args.url or os.environ.get("OPSIOM_URL")
|
|
330
|
+
url = normalize_url(url_override or config.get("url"))
|
|
331
|
+
# Le type de serveur enregistré ne vaut que pour l'URL enregistrée : si
|
|
332
|
+
# l'URL vient d'une option ou d'une variable d'environnement, on le redétecte.
|
|
333
|
+
backend_pref = args.backend or os.environ.get("OPSIOM_BACKEND") or (
|
|
334
|
+
"auto" if url_override else config.get("backend", "auto"))
|
|
274
335
|
api_key = args.api_key or os.environ.get("OPSIOM_API_KEY") or config.get("api_key")
|
|
336
|
+
username = args.username or os.environ.get("OPSIOM_USERNAME") or config.get("username")
|
|
275
337
|
if args.configure or not url:
|
|
276
|
-
new_config = prompt_for_config()
|
|
338
|
+
new_config = prompt_for_config(backend_pref)
|
|
277
339
|
url = new_config["url"] or url
|
|
340
|
+
backend_pref = new_config["backend"]
|
|
278
341
|
api_key = new_config["api_key"] or api_key
|
|
279
|
-
|
|
342
|
+
username = new_config["username"] or username
|
|
343
|
+
return {
|
|
344
|
+
"url": url,
|
|
345
|
+
"backend": detect_backend(url, backend_pref),
|
|
346
|
+
"api_key": api_key,
|
|
347
|
+
"username": username,
|
|
348
|
+
"hf_token": args.hf_token or os.environ.get("OPSIOM_HF_TOKEN") or os.environ.get("HF_TOKEN"),
|
|
349
|
+
}
|
|
280
350
|
|
|
281
351
|
|
|
282
352
|
# ----------------------------------------------------------------------------
|
|
283
|
-
# Quota — 500 tokens/jour par clé API
|
|
284
|
-
#
|
|
285
|
-
#
|
|
286
|
-
#
|
|
287
|
-
#
|
|
288
|
-
#
|
|
289
|
-
#
|
|
353
|
+
# Quota — 500 tokens/jour par clé API ; sur un Space Hugging Face la limite
|
|
354
|
+
# dépend du forfait du compte (elle est donc lue dans la réponse du serveur,
|
|
355
|
+
# jamais supposée). Le fichier usage.json ne sert plus qu'à AFFICHER un dernier
|
|
356
|
+
# quota connu avant la toute première requête de la session (utile pour
|
|
357
|
+
# /quota juste après le lancement) : dès qu'une réponse arrive, c'est le
|
|
358
|
+
# quota renvoyé par le serveur ("remaining_quota" pour le serveur API,
|
|
359
|
+
# objet "quota" pour un Space) qui fait foi et remplace la valeur locale. Un
|
|
360
|
+
# utilisateur qui modifierait usage.json à la main ne gagnerait donc plus
|
|
361
|
+
# rien : le serveur refuse les requêtes une fois son propre compteur à zéro (429).
|
|
362
|
+
# Les entrées sont indexées par "identité" : empreinte de la clé API, ou du
|
|
363
|
+
# couple (Space, pseudo) pour un compte Octix.
|
|
290
364
|
# ----------------------------------------------------------------------------
|
|
291
365
|
|
|
292
366
|
def _api_key_id(api_key):
|
|
@@ -298,6 +372,12 @@ def _api_key_id(api_key):
|
|
|
298
372
|
return hashlib.sha256(api_key.encode("utf-8")).hexdigest()[:16]
|
|
299
373
|
|
|
300
374
|
|
|
375
|
+
def _account_id(url, username):
|
|
376
|
+
"""Identité d'un compte Octix sur un Space (empreinte, pas de clair)."""
|
|
377
|
+
raw = f"{url.rstrip('/').lower()}|{(username or '').lower()}"
|
|
378
|
+
return "hf-" + hashlib.sha256(raw.encode("utf-8")).hexdigest()[:16]
|
|
379
|
+
|
|
380
|
+
|
|
301
381
|
def today_str():
|
|
302
382
|
return time.strftime("%Y-%m-%d")
|
|
303
383
|
|
|
@@ -323,37 +403,89 @@ def save_usage(usage):
|
|
|
323
403
|
USAGE_PATH.write_text(json.dumps(usage, indent=2))
|
|
324
404
|
|
|
325
405
|
|
|
326
|
-
def get_quota_state(
|
|
327
|
-
"""
|
|
328
|
-
réinitialise tout seul dès que la date enregistrée n'est plus
|
|
329
|
-
du jour -- pas besoin de tâche planifiée.
|
|
330
|
-
|
|
406
|
+
def get_quota_state(identity):
|
|
407
|
+
"""{'used', 'limit', 'remaining', 'unlimited'} pour aujourd'hui. Le
|
|
408
|
+
compteur se réinitialise tout seul dès que la date enregistrée n'est plus
|
|
409
|
+
celle du jour -- pas besoin de tâche planifiée. Pour un compte sans
|
|
410
|
+
limite, 'remaining' vaut None."""
|
|
411
|
+
entry = load_usage().get(identity, {})
|
|
331
412
|
if entry.get("date") != today_str():
|
|
332
|
-
return 0, DAILY_TOKEN_QUOTA
|
|
413
|
+
return {"used": 0, "limit": DAILY_TOKEN_QUOTA, "remaining": DAILY_TOKEN_QUOTA, "unlimited": False}
|
|
333
414
|
used = entry.get("tokens_used", 0)
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
415
|
+
if entry.get("unlimited"):
|
|
416
|
+
return {"used": used, "limit": 0, "remaining": None, "unlimited": True}
|
|
417
|
+
limit = entry.get("limit", DAILY_TOKEN_QUOTA)
|
|
418
|
+
return {"used": used, "limit": limit, "remaining": max(0, limit - used), "unlimited": False}
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def record_server_quota(identity, remaining=None, used=None, limit=None, unlimited=False):
|
|
422
|
+
"""Remplace le compteur local par le quota renvoyé par le serveur (lui-même
|
|
423
|
+
décompté côté Octix) — c'est la seule valeur qui compte, le fichier local
|
|
424
|
+
n'est qu'un cache d'affichage. Un Space donne (used, limit) ; le serveur
|
|
425
|
+
API ne donne que 'remaining', pour une limite fixe de DAILY_TOKEN_QUOTA."""
|
|
426
|
+
if unlimited:
|
|
427
|
+
entry = {"date": today_str(), "tokens_used": used or 0, "unlimited": True}
|
|
428
|
+
elif used is not None and limit is not None:
|
|
429
|
+
entry = {"date": today_str(), "tokens_used": used, "limit": limit}
|
|
430
|
+
elif remaining is not None:
|
|
431
|
+
entry = {"date": today_str(), "tokens_used": max(0, DAILY_TOKEN_QUOTA - remaining)}
|
|
432
|
+
else:
|
|
342
433
|
return
|
|
343
434
|
usage = load_usage()
|
|
344
|
-
|
|
345
|
-
usage[key_id] = {
|
|
346
|
-
"date": today_str(),
|
|
347
|
-
"tokens_used": max(0, DAILY_TOKEN_QUOTA - remaining),
|
|
348
|
-
}
|
|
435
|
+
usage[identity] = entry
|
|
349
436
|
save_usage(usage)
|
|
350
437
|
|
|
351
438
|
|
|
352
439
|
# ----------------------------------------------------------------------------
|
|
353
|
-
#
|
|
440
|
+
# Clients — un par type de serveur, avec la même interface pour le chat :
|
|
441
|
+
# health() -> {"status": "ok"|..., "models_loaded": [...], ...}
|
|
442
|
+
# list_models() -> {"models": [{"id","label","params"[,"locked"]}], "default"}
|
|
443
|
+
# chat_stream() -> générateur : fragments de texte, puis le dict final
|
|
444
|
+
# Attributs communs : backend, base_url, identity (clé du cache de quota),
|
|
445
|
+
# auth_label, auth_error, tracks_quota.
|
|
354
446
|
# ----------------------------------------------------------------------------
|
|
355
447
|
|
|
448
|
+
class ServerError(Exception):
|
|
449
|
+
"""Erreur signalée par le serveur DANS le flux SSE (après un statut 200)."""
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def server_error_message(http_error):
|
|
453
|
+
"""Message d'erreur JSON ({"error": "..."}) d'une réponse HTTP, ou None."""
|
|
454
|
+
try:
|
|
455
|
+
return http_error.response.json().get("error")
|
|
456
|
+
except (ValueError, AttributeError):
|
|
457
|
+
return None
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def stream_pieces(response):
|
|
461
|
+
"""Décode un flux SSE ("data: {...}") : cède chaque fragment de texte, puis
|
|
462
|
+
le dict final (évènement "done"). Lève ServerError sur un évènement
|
|
463
|
+
{"error": "..."}."""
|
|
464
|
+
response.encoding = "utf-8"
|
|
465
|
+
for line in response.iter_lines(decode_unicode=True):
|
|
466
|
+
if not line or not line.startswith("data: "):
|
|
467
|
+
continue
|
|
468
|
+
try:
|
|
469
|
+
event = json.loads(line[len("data: "):])
|
|
470
|
+
except ValueError:
|
|
471
|
+
continue
|
|
472
|
+
if event.get("error"):
|
|
473
|
+
raise ServerError(event["error"])
|
|
474
|
+
if event.get("done"):
|
|
475
|
+
yield event
|
|
476
|
+
return
|
|
477
|
+
token = event.get("token")
|
|
478
|
+
if token:
|
|
479
|
+
yield token
|
|
480
|
+
|
|
481
|
+
|
|
356
482
|
class OpsiomClient:
|
|
483
|
+
"""Serveur d'inférence classique (ngrok, Render…) : routes /api/*,
|
|
484
|
+
authentification par clé API (X-API-Key)."""
|
|
485
|
+
|
|
486
|
+
backend = "api"
|
|
487
|
+
auth_error = ("clé API invalide ou expirée.", "reconfigure avec : opsiom --configure")
|
|
488
|
+
|
|
357
489
|
def __init__(self, base_url, api_key=None, timeout=60):
|
|
358
490
|
self.base_url = base_url.rstrip("/")
|
|
359
491
|
self.timeout = timeout
|
|
@@ -362,6 +494,12 @@ class OpsiomClient:
|
|
|
362
494
|
if api_key:
|
|
363
495
|
headers["X-API-Key"] = api_key
|
|
364
496
|
self.session.headers.update(headers)
|
|
497
|
+
self.identity = _api_key_id(api_key)
|
|
498
|
+
self.auth_label = "clé API" if api_key else "aucune"
|
|
499
|
+
self.tracks_quota = bool(api_key)
|
|
500
|
+
|
|
501
|
+
def new_conversation(self):
|
|
502
|
+
return False # le serveur ne garde aucun historique
|
|
365
503
|
|
|
366
504
|
def health(self):
|
|
367
505
|
r = self.session.get(f"{self.base_url}/api/health", timeout=10)
|
|
@@ -398,16 +536,118 @@ class OpsiomClient:
|
|
|
398
536
|
timeout=self.timeout, stream=True,
|
|
399
537
|
)
|
|
400
538
|
r.raise_for_status()
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
539
|
+
yield from stream_pieces(r)
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
class HFSpaceClient:
|
|
543
|
+
"""Space Hugging Face qui héberge l'interface Opsiom. Le Space n'a pas de
|
|
544
|
+
clé API : il authentifie un COMPTE Octix par une session (cookie) ouverte
|
|
545
|
+
via son formulaire /login. Routes : /status, /models, /chat/stream.
|
|
546
|
+
Le quota (forfait, codes de promo) et l'historique des conversations
|
|
547
|
+
sont tenus côté Space, par compte -- le même qu'à l'usage sur le web."""
|
|
548
|
+
|
|
549
|
+
backend = "hf"
|
|
550
|
+
tracks_quota = True
|
|
551
|
+
auth_error = ("session expirée ou refusée par le Space.", None)
|
|
552
|
+
|
|
553
|
+
def __init__(self, base_url, username=None, hf_token=None, timeout=120):
|
|
554
|
+
self.base_url = base_url.rstrip("/")
|
|
555
|
+
self.timeout = timeout # CPU partagé : le premier fragment peut tarder
|
|
556
|
+
self.username = username
|
|
557
|
+
self.identity = _account_id(self.base_url, username)
|
|
558
|
+
self.conversation_id = None # historique serveur : une conversation par session
|
|
559
|
+
self.session = requests.Session()
|
|
560
|
+
if hf_token: # Space privé : jeton d'accès Hugging Face
|
|
561
|
+
self.session.headers["Authorization"] = f"Bearer {hf_token}"
|
|
562
|
+
|
|
563
|
+
@property
|
|
564
|
+
def auth_label(self):
|
|
565
|
+
return f"compte Octix ({self.username})" if self.username else "compte Octix"
|
|
566
|
+
|
|
567
|
+
def new_conversation(self):
|
|
568
|
+
self.conversation_id = None
|
|
569
|
+
return True
|
|
570
|
+
|
|
571
|
+
def login(self, username, password):
|
|
572
|
+
"""Ouvre la session Octix du Space (même formulaire que la page /login).
|
|
573
|
+
Renvoie (True, None) ou (False, message)."""
|
|
574
|
+
try:
|
|
575
|
+
r = self.session.post(
|
|
576
|
+
f"{self.base_url}/login",
|
|
577
|
+
data={"username": username, "password": password},
|
|
578
|
+
timeout=90, # un Space endormi met un moment à se réveiller
|
|
579
|
+
allow_redirects=False,
|
|
580
|
+
)
|
|
581
|
+
except requests.RequestException as e:
|
|
582
|
+
return False, f"Space injoignable : {e}"
|
|
583
|
+
if r.status_code in (301, 302, 303, 307, 308):
|
|
584
|
+
self.username = username
|
|
585
|
+
self.identity = _account_id(self.base_url, username)
|
|
586
|
+
return True, None
|
|
587
|
+
flash = re.search(r'<li class="flash[^"]*">(.*?)</li>', r.text or "", re.S)
|
|
588
|
+
if flash:
|
|
589
|
+
return False, html.unescape(flash.group(1)).strip()
|
|
590
|
+
if r.status_code in (400, 401):
|
|
591
|
+
return False, "pseudo ou mot de passe refusé."
|
|
592
|
+
return False, (f"réponse inattendue du serveur (HTTP {r.status_code}) — "
|
|
593
|
+
"le Space est peut-être en train de démarrer, réessaie dans une minute.")
|
|
594
|
+
|
|
595
|
+
def health(self):
|
|
596
|
+
r = self.session.get(f"{self.base_url}/status", timeout=30)
|
|
597
|
+
r.raise_for_status()
|
|
598
|
+
data = r.json()
|
|
599
|
+
return {
|
|
600
|
+
"status": "ok" if data.get("online") else (data.get("error") or "indisponible"),
|
|
601
|
+
"models_loaded": data.get("models_loaded", []),
|
|
602
|
+
"message": data.get("message"),
|
|
603
|
+
"quota": data.get("quota"),
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
def list_models(self):
|
|
607
|
+
r = self.session.get(f"{self.base_url}/models", timeout=30)
|
|
608
|
+
r.raise_for_status()
|
|
609
|
+
return r.json()
|
|
610
|
+
|
|
611
|
+
def chat_stream(self, message, model=None, **params):
|
|
612
|
+
"""Même contrat que OpsiomClient.chat_stream, avec en plus le suivi de
|
|
613
|
+
la conversation (sinon chaque message créerait une conversation dans
|
|
614
|
+
l'historique du compte). Le dict final porte 'quota' (used/limit/...)."""
|
|
615
|
+
payload = {"message": message}
|
|
616
|
+
if model:
|
|
617
|
+
payload["model"] = model
|
|
618
|
+
if self.conversation_id:
|
|
619
|
+
payload["conversation_id"] = self.conversation_id
|
|
620
|
+
payload.update(params)
|
|
621
|
+
r = self.session.post(
|
|
622
|
+
f"{self.base_url}/chat/stream", json=payload,
|
|
623
|
+
timeout=self.timeout, stream=True,
|
|
624
|
+
)
|
|
625
|
+
r.raise_for_status()
|
|
626
|
+
for piece in stream_pieces(r):
|
|
627
|
+
if isinstance(piece, dict):
|
|
628
|
+
self.conversation_id = piece.get("conversation_id") or self.conversation_id
|
|
629
|
+
yield piece
|
|
630
|
+
|
|
631
|
+
|
|
632
|
+
def login_interactive(client, username=None, attempts=3):
|
|
633
|
+
"""Demande le mot de passe (jamais affiché ni enregistré) et ouvre la
|
|
634
|
+
session. OPSIOM_PASSWORD, s'il est défini, est essayé en premier."""
|
|
635
|
+
try:
|
|
636
|
+
username = username or input(" Pseudo Octix : ").strip()
|
|
637
|
+
password = os.environ.get("OPSIOM_PASSWORD")
|
|
638
|
+
for _ in range(attempts):
|
|
639
|
+
if not password:
|
|
640
|
+
password = getpass.getpass(f" Mot de passe Octix de {username} : ")
|
|
641
|
+
print(muted(" connexion… "), end="\r", flush=True)
|
|
642
|
+
ok, message = client.login(username, password)
|
|
643
|
+
print("\033[K", end="")
|
|
644
|
+
if ok:
|
|
645
|
+
return True
|
|
646
|
+
print_error("connexion refusée.", message)
|
|
647
|
+
password = None
|
|
648
|
+
except (EOFError, KeyboardInterrupt):
|
|
649
|
+
print()
|
|
650
|
+
return False
|
|
411
651
|
|
|
412
652
|
|
|
413
653
|
# ----------------------------------------------------------------------------
|
|
@@ -421,6 +661,7 @@ HELP_TEXT = """
|
|
|
421
661
|
/status réaffiche l'en-tête de connexion
|
|
422
662
|
/quota affiche le quota de tokens restant aujourd'hui
|
|
423
663
|
/save [nom] exporte la conversation en Markdown
|
|
664
|
+
/new nouvelle conversation (Space Hugging Face : nouvel historique)
|
|
424
665
|
/clear efface l'écran
|
|
425
666
|
/help affiche cette aide
|
|
426
667
|
/quit, /exit quitte le CLI
|
|
@@ -453,6 +694,8 @@ def print_models(models_payload, current_model):
|
|
|
453
694
|
marker = accent("›") if is_current else " "
|
|
454
695
|
label = accent(m["label"], bold=True) if is_current else m["label"]
|
|
455
696
|
detail = f"{m['params']} · {m['id']}"
|
|
697
|
+
if m.get("locked"):
|
|
698
|
+
detail += " · non inclus dans ton forfait"
|
|
456
699
|
print(f" {marker} {muted(f'[{i}]')} {label} {muted(detail)}")
|
|
457
700
|
|
|
458
701
|
|
|
@@ -469,6 +712,10 @@ def resolve_model_choice(choice, models):
|
|
|
469
712
|
return matches[0] if len(matches) == 1 else None
|
|
470
713
|
|
|
471
714
|
|
|
715
|
+
def model_is_locked(models, model_id):
|
|
716
|
+
return any(m["id"] == model_id and m.get("locked") for m in models)
|
|
717
|
+
|
|
718
|
+
|
|
472
719
|
def select_model_interactively(models_payload, current_model):
|
|
473
720
|
models = models_payload.get("models", [])
|
|
474
721
|
if not models:
|
|
@@ -482,6 +729,9 @@ def select_model_interactively(models_payload, current_model):
|
|
|
482
729
|
if resolved is None:
|
|
483
730
|
print_error(f"choix invalide : '{choice}'")
|
|
484
731
|
return current_model
|
|
732
|
+
if model_is_locked(models, resolved):
|
|
733
|
+
print_error(f"le modèle « {resolved} » n'est pas inclus dans ton forfait.")
|
|
734
|
+
return current_model
|
|
485
735
|
print(f" {accent('→')} modèle actif : {accent(resolved, bold=True)}")
|
|
486
736
|
return resolved
|
|
487
737
|
|
|
@@ -541,19 +791,26 @@ class StreamPrinter:
|
|
|
541
791
|
print()
|
|
542
792
|
|
|
543
793
|
|
|
544
|
-
def run_chat(client
|
|
794
|
+
def run_chat(client):
|
|
545
795
|
setup_history()
|
|
546
796
|
transcript = []
|
|
547
797
|
current_model = None
|
|
548
798
|
models_payload = {"models": []}
|
|
549
799
|
connected = False
|
|
550
800
|
models_loaded = []
|
|
801
|
+
server_notice = None
|
|
551
802
|
|
|
552
803
|
invalid_key = False
|
|
553
804
|
try:
|
|
554
805
|
health = client.health()
|
|
555
806
|
connected = health.get("status") == "ok"
|
|
556
807
|
models_loaded = health.get("models_loaded", [])
|
|
808
|
+
if not connected:
|
|
809
|
+
server_notice = health.get("message")
|
|
810
|
+
quota = health.get("quota")
|
|
811
|
+
if isinstance(quota, dict):
|
|
812
|
+
record_server_quota(client.identity, remaining=quota.get("remaining"), used=quota.get("used"),
|
|
813
|
+
limit=quota.get("limit"), unlimited=bool(quota.get("unlimited")))
|
|
557
814
|
except requests.HTTPError as e:
|
|
558
815
|
connected = False
|
|
559
816
|
invalid_key = e.response is not None and e.response.status_code in (401, 403)
|
|
@@ -567,9 +824,11 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
567
824
|
pass
|
|
568
825
|
|
|
569
826
|
if invalid_key:
|
|
570
|
-
print_error(
|
|
827
|
+
print_error(*client.auth_error)
|
|
828
|
+
elif server_notice:
|
|
829
|
+
print_error(server_notice)
|
|
571
830
|
|
|
572
|
-
print_header(
|
|
831
|
+
print_header(client, connected, models_loaded, current_model)
|
|
573
832
|
print(f" {muted('Tapez votre message, ou /help pour la liste des commandes.')} ")
|
|
574
833
|
|
|
575
834
|
while True:
|
|
@@ -595,10 +854,19 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
595
854
|
sys.stdout.write("\033[H\033[J")
|
|
596
855
|
sys.stdout.flush()
|
|
597
856
|
elif cmd == "status":
|
|
598
|
-
print_header(
|
|
857
|
+
print_header(client, connected, models_loaded, current_model, show_mark=False)
|
|
599
858
|
elif cmd == "quota":
|
|
600
|
-
|
|
601
|
-
|
|
859
|
+
state = get_quota_state(client.identity)
|
|
860
|
+
if state["unlimited"]:
|
|
861
|
+
print(f" {muted('quota illimité sur ce compte.')}")
|
|
862
|
+
else:
|
|
863
|
+
text = f"{state['used']}/{state['limit']} tokens utilisés aujourd’hui — {state['remaining']} restants."
|
|
864
|
+
print(f" {muted(text)}")
|
|
865
|
+
elif cmd == "new":
|
|
866
|
+
if client.new_conversation():
|
|
867
|
+
print(f" {accent('→')} nouvelle conversation")
|
|
868
|
+
else:
|
|
869
|
+
print(muted(" ce serveur ne garde pas d'historique : chaque message est indépendant."))
|
|
602
870
|
elif cmd == "save":
|
|
603
871
|
if not transcript:
|
|
604
872
|
print(muted(" rien à enregistrer pour l'instant."))
|
|
@@ -616,17 +884,23 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
616
884
|
current_model = select_model_interactively(models_payload, current_model)
|
|
617
885
|
else:
|
|
618
886
|
resolved = resolve_model_choice(arg, models_payload.get("models", [])) or arg
|
|
619
|
-
|
|
620
|
-
|
|
887
|
+
if model_is_locked(models_payload.get("models", []), resolved):
|
|
888
|
+
print_error(f"le modèle « {resolved} » n'est pas inclus dans ton forfait.")
|
|
889
|
+
else:
|
|
890
|
+
current_model = resolved
|
|
891
|
+
print(f" {accent('→')} modèle actif : {accent(resolved, bold=True)}")
|
|
621
892
|
else:
|
|
622
893
|
print_error(f"commande inconnue : /{cmd}")
|
|
623
894
|
print()
|
|
624
895
|
continue
|
|
625
896
|
|
|
626
|
-
|
|
627
|
-
|
|
897
|
+
# Serveur API : garde-fou local. Space : le quota (forfait, codes de promo)
|
|
898
|
+
# change côté serveur sans que le CLI le sache -- on laisse donc le
|
|
899
|
+
# Space trancher (réponse 429 avec son propre message).
|
|
900
|
+
state = get_quota_state(client.identity)
|
|
901
|
+
if client.backend == "api" and quota_exhausted(state):
|
|
628
902
|
print_error(
|
|
629
|
-
f"quota quotidien de {
|
|
903
|
+
f"quota quotidien de {state['limit']} tokens atteint.",
|
|
630
904
|
"réessaie demain, ou utilise une autre clé API avec --configure",
|
|
631
905
|
)
|
|
632
906
|
print()
|
|
@@ -656,18 +930,28 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
656
930
|
spinner.stop()
|
|
657
931
|
print(muted("\n (requête annulée) "))
|
|
658
932
|
continue
|
|
933
|
+
except ServerError as e:
|
|
934
|
+
if spinner_running:
|
|
935
|
+
spinner.stop()
|
|
936
|
+
elif response_parts:
|
|
937
|
+
printer.close() # termine proprement la ligne en cours
|
|
938
|
+
print_error(str(e))
|
|
939
|
+
print()
|
|
940
|
+
continue
|
|
659
941
|
except requests.HTTPError as e:
|
|
660
942
|
if spinner_running:
|
|
661
943
|
spinner.stop()
|
|
662
|
-
if e.response is not None
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
pass
|
|
944
|
+
status = e.response.status_code if e.response is not None else None
|
|
945
|
+
detail = server_error_message(e)
|
|
946
|
+
if status == 401 or (status == 403 and client.backend == "api"):
|
|
947
|
+
print_error(*client.auth_error)
|
|
948
|
+
if client.backend == "hf" and login_interactive(client, client.username, attempts=1):
|
|
949
|
+
print(muted(" reconnecté — renvoie ton message."))
|
|
950
|
+
elif status == 429:
|
|
670
951
|
print_error("quota dépassé côté serveur.", detail or "réessaie plus tard")
|
|
952
|
+
elif detail:
|
|
953
|
+
# p. ex. 403 « modèle non inclus dans ton forfait », 413, 503 « modèles en chargement »
|
|
954
|
+
print_error(detail)
|
|
671
955
|
else:
|
|
672
956
|
print_error("la requête a échoué ", str(e))
|
|
673
957
|
print()
|
|
@@ -688,24 +972,32 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
688
972
|
used_model = final_event.get("model", current_model or "?")
|
|
689
973
|
print(f" {muted(format_elapsed(elapsed_ms))}")
|
|
690
974
|
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
975
|
+
server_quota = final_event.get("quota") # Space : {used, limit, remaining, ...}
|
|
976
|
+
remaining_quota = final_event.get("remaining_quota") # serveur API
|
|
977
|
+
if isinstance(server_quota, dict):
|
|
978
|
+
record_server_quota(client.identity, remaining=server_quota.get("remaining"),
|
|
979
|
+
used=server_quota.get("used"), limit=server_quota.get("limit"),
|
|
980
|
+
unlimited=bool(server_quota.get("unlimited")))
|
|
981
|
+
elif remaining_quota is not None:
|
|
982
|
+
record_server_quota(client.identity, remaining=remaining_quota)
|
|
694
983
|
else:
|
|
695
984
|
# Le serveur n'a pas renvoyé de quota faisant autorité (Octix
|
|
696
985
|
# injoignable côté serveur, p. ex.) : on retombe sur l'estimation
|
|
697
986
|
# locale plutôt que de ne rien afficher.
|
|
698
987
|
estimated = estimate_tokens(user_input) + estimate_tokens(response_text)
|
|
699
988
|
usage = load_usage()
|
|
700
|
-
|
|
701
|
-
entry = usage.get(key_id, {})
|
|
989
|
+
entry = usage.get(client.identity, {})
|
|
702
990
|
if entry.get("date") != today_str():
|
|
703
991
|
entry = {"date": today_str(), "tokens_used": 0}
|
|
704
992
|
entry["tokens_used"] = entry.get("tokens_used", 0) + estimated
|
|
705
|
-
usage[
|
|
993
|
+
usage[client.identity] = entry
|
|
706
994
|
save_usage(usage)
|
|
707
|
-
|
|
708
|
-
|
|
995
|
+
state = get_quota_state(client.identity)
|
|
996
|
+
if state["unlimited"]:
|
|
997
|
+
print(f" {muted('quota illimité')}")
|
|
998
|
+
else:
|
|
999
|
+
text = f"{state['remaining']} tokens restants aujourd’hui"
|
|
1000
|
+
print(f" {muted(text)}")
|
|
709
1001
|
print()
|
|
710
1002
|
transcript.append({"role": "assistant", "text": response_text, "model": used_model, "elapsed_ms": elapsed_ms})
|
|
711
1003
|
|
|
@@ -718,25 +1010,35 @@ def run_chat(client: OpsiomClient, url, api_key):
|
|
|
718
1010
|
|
|
719
1011
|
def main():
|
|
720
1012
|
parser = argparse.ArgumentParser(prog="opsiom", description="CLI de chat pour Opsiom")
|
|
721
|
-
parser.add_argument("--url", default=None, help="URL du serveur Opsiom (sinon config/prompt)")
|
|
722
|
-
parser.add_argument("--api-key", default=None, help="clé API si le serveur en exige une")
|
|
1013
|
+
parser.add_argument("--url", default=None, help="URL du serveur Opsiom ou du Space Hugging Face (sinon config/prompt)")
|
|
1014
|
+
parser.add_argument("--api-key", default=None, help="clé API si le serveur en exige une (serveur API)")
|
|
1015
|
+
parser.add_argument("--backend", choices=("auto", "api", "hf"), default=None,
|
|
1016
|
+
help="type de serveur : api (ngrok…) ou hf (Space Hugging Face) ; auto = d'après l'URL")
|
|
1017
|
+
parser.add_argument("--username", default=None, help="pseudo Octix (Space Hugging Face)")
|
|
1018
|
+
parser.add_argument("--hf-token", default=None,
|
|
1019
|
+
help="jeton d'accès Hugging Face, si le Space est privé (ou variable HF_TOKEN)")
|
|
723
1020
|
parser.add_argument("--configure", action="store_true", help="reconfigure l'URL/clé et les réenregistre")
|
|
724
1021
|
parser.add_argument("--no-logo", action="store_true", help="masque le repère ASCII au démarrage")
|
|
725
1022
|
parser.add_argument("--version", action="version", version=f"opsiom-cli {VERSION}")
|
|
726
1023
|
args = parser.parse_args()
|
|
727
1024
|
|
|
728
1025
|
|
|
729
|
-
|
|
730
|
-
if not url:
|
|
1026
|
+
settings = resolve_settings(args)
|
|
1027
|
+
if not settings["url"]:
|
|
731
1028
|
print_error("aucune URL configurée, abandon.")
|
|
732
1029
|
sys.exit(1)
|
|
733
1030
|
|
|
734
|
-
|
|
1031
|
+
if settings["backend"] == "hf":
|
|
1032
|
+
client = HFSpaceClient(settings["url"], username=settings["username"], hf_token=settings["hf_token"])
|
|
1033
|
+
if not login_interactive(client, settings["username"]):
|
|
1034
|
+
sys.exit(1)
|
|
1035
|
+
else:
|
|
1036
|
+
client = OpsiomClient(settings["url"], api_key=settings["api_key"])
|
|
735
1037
|
if args.no_logo:
|
|
736
1038
|
global OCTOPUS_MARK
|
|
737
1039
|
OCTOPUS_MARK = []
|
|
738
1040
|
|
|
739
|
-
run_chat(client
|
|
1041
|
+
run_chat(client)
|
|
740
1042
|
|
|
741
1043
|
|
|
742
1044
|
if __name__ == "__main__":
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|