hugpy-core 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hugpy_core-0.1.0/PKG-INFO +32 -0
- hugpy_core-0.1.0/README.md +9 -0
- hugpy_core-0.1.0/hugpy_core/__init__.py +3 -0
- hugpy_core-0.1.0/hugpy_core/app.py +30 -0
- hugpy_core-0.1.0/hugpy_core/console_routes.py +155 -0
- hugpy_core-0.1.0/hugpy_core/keys.py +62 -0
- hugpy_core-0.1.0/hugpy_core/live.py +127 -0
- hugpy_core-0.1.0/hugpy_core/modeldb.py +283 -0
- hugpy_core-0.1.0/hugpy_core/openai.py +70 -0
- hugpy_core-0.1.0/hugpy_core/pick.py +87 -0
- hugpy_core-0.1.0/hugpy_core/planner.py +303 -0
- hugpy_core-0.1.0/hugpy_core/registry.py +170 -0
- hugpy_core-0.1.0/hugpy_core/relay.py +110 -0
- hugpy_core-0.1.0/hugpy_core/routes.py +510 -0
- hugpy_core-0.1.0/hugpy_core/testfire.py +838 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/PKG-INFO +32 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/SOURCES.txt +21 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/dependency_links.txt +1 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/entry_points.txt +2 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/requires.txt +6 -0
- hugpy_core-0.1.0/hugpy_core.egg-info/top_level.txt +1 -0
- hugpy_core-0.1.0/pyproject.toml +36 -0
- hugpy_core-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hugpy-core
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: hugpy central, minimal: live worker registry (register/heartbeat), model->worker pick, and the OpenAI /v1 relay to workers. Lifted from legacy hugpy central without evict-to-fit, discovery, admission or DB state.
|
|
5
|
+
Author-email: putkoff <support@hugpy.ai>
|
|
6
|
+
License: Proprietary
|
|
7
|
+
Project-URL: Homepage, https://hugpy.ai
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Environment :: Web Environment
|
|
10
|
+
Classifier: Framework :: Flask
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
Requires-Dist: flask>=2
|
|
19
|
+
Requires-Dist: abstract_flask
|
|
20
|
+
Requires-Dist: psycopg[binary]>=3
|
|
21
|
+
Provides-Extra: test
|
|
22
|
+
Requires-Dist: pytest; extra == "test"
|
|
23
|
+
|
|
24
|
+
# hugpy_core
|
|
25
|
+
|
|
26
|
+
hugpy central, minimal. Workers (`hugpy_worker`, one per box, over fitevict)
|
|
27
|
+
register and heartbeat here; the registry is live in memory. `/v1/chat/completions`
|
|
28
|
+
picks an online worker that has the model (resident first; `alloc.worker` pins one)
|
|
29
|
+
and relays to its `/infer/stream`. No evict-to-fit, discovery, admission or DB
|
|
30
|
+
state — each box's fitevict owns placement and the call log.
|
|
31
|
+
|
|
32
|
+
hugpy-core --host 127.0.0.1 --port 7020
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# hugpy_core
|
|
2
|
+
|
|
3
|
+
hugpy central, minimal. Workers (`hugpy_worker`, one per box, over fitevict)
|
|
4
|
+
register and heartbeat here; the registry is live in memory. `/v1/chat/completions`
|
|
5
|
+
picks an online worker that has the model (resident first; `alloc.worker` pins one)
|
|
6
|
+
and relays to its `/infer/stream`. No evict-to-fit, discovery, admission or DB
|
|
7
|
+
state — each box's fitevict owns placement and the call log.
|
|
8
|
+
|
|
9
|
+
hugpy-core --host 127.0.0.1 --port 7020
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""hugpy_core router app: abstract_flask app over the routes module."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import logging
|
|
6
|
+
|
|
7
|
+
from abstract_flask import get_Flask_app
|
|
8
|
+
|
|
9
|
+
from . import routes
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger("hugpy_core")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def build_app():
|
|
15
|
+
return get_Flask_app(name="hugpy_core", routes=routes)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def main(argv=None) -> int:
|
|
19
|
+
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
|
20
|
+
ap = argparse.ArgumentParser(prog="hugpy_core")
|
|
21
|
+
ap.add_argument("--host", default="127.0.0.1")
|
|
22
|
+
ap.add_argument("--port", type=int, default=7020)
|
|
23
|
+
a = ap.parse_args(argv)
|
|
24
|
+
logger.info("hugpy_core router on %s:%s", a.host, a.port)
|
|
25
|
+
build_app().run(host=a.host, port=a.port, threaded=True)
|
|
26
|
+
return 0
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
if __name__ == "__main__":
|
|
30
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""The console's remaining routes, served by hugpy_core itself (no fallback to
|
|
2
|
+
the hugpy_next central). Live values come from the DB / the registry at the
|
|
3
|
+
moment of the call; /llm/events pushes on every registry change."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
import time
|
|
8
|
+
import urllib.request
|
|
9
|
+
|
|
10
|
+
from flask import Blueprint, Response, jsonify, request, stream_with_context
|
|
11
|
+
|
|
12
|
+
from . import registry
|
|
13
|
+
|
|
14
|
+
IE_DSN = "dbname=inference_engine"
|
|
15
|
+
NOTE_FLAGS = ["broken", "trash", "archive", "unknown", "needs-env", "experimental", "keep"]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _ie(sql: str, params: tuple = ()) -> list:
|
|
19
|
+
import psycopg
|
|
20
|
+
with psycopg.connect(IE_DSN, autocommit=True, connect_timeout=5) as c:
|
|
21
|
+
return c.execute(sql, params).fetchall()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def worker_activity(worker: dict) -> list:
|
|
25
|
+
"""One host view of the worker's GPU: every process on its card from
|
|
26
|
+
inference_engine.gpu_state (published by its fitevict front door the moment
|
|
27
|
+
the card changes), joined to the front door's residents."""
|
|
28
|
+
name = worker.get("name")
|
|
29
|
+
cards = _ie("SELECT device, processes, hugpy_pids, probe_pids FROM gpu_state WHERE worker = %s", (name,))
|
|
30
|
+
res = {int(pid): key for key, pid in _ie("SELECT key, pid FROM residents WHERE worker = %s AND pid IS NOT NULL", (name,))}
|
|
31
|
+
rows = []
|
|
32
|
+
for device, procs, ours, probes in cards:
|
|
33
|
+
for pid_s, nbytes in sorted((procs or {}).items()):
|
|
34
|
+
pid = int(pid_s)
|
|
35
|
+
key = res.get(pid)
|
|
36
|
+
kind = "hugpy" if key or pid in (ours or []) else ("probe" if pid in (probes or []) else "observed")
|
|
37
|
+
rows.append({"kind": kind, "model_key": key, "service": "fitevict" if kind == "hugpy" else
|
|
38
|
+
("llama.cpp dry run" if kind == "probe" else "gpu process"),
|
|
39
|
+
"pids": [pid], "gpu_indices": [device], "vram_bytes": int(nbytes),
|
|
40
|
+
"vram_mib": int(nbytes) // 1048576, "state": "running",
|
|
41
|
+
"evictable": kind == "hugpy", "immutable": kind == "observed",
|
|
42
|
+
"api_available": bool(key), "activity": {},
|
|
43
|
+
"note": ("Hugpy managed (fitevict)" if kind == "hugpy" else
|
|
44
|
+
"transient sizing probe" if kind == "probe" else
|
|
45
|
+
"GPU process observed; not managed by hugpy")})
|
|
46
|
+
return rows
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def attach_console(bp: Blueprint, snapshot_payload) -> None:
|
|
50
|
+
|
|
51
|
+
@bp.get("/llm/workers/<worker_id>/activity")
|
|
52
|
+
def workers_activity(worker_id):
|
|
53
|
+
w = registry.lookup_worker(worker_id)
|
|
54
|
+
if w is None:
|
|
55
|
+
return jsonify({"ok": False, "error": f"no worker {worker_id!r}", "activity": []}), 404
|
|
56
|
+
try:
|
|
57
|
+
return jsonify({"ok": True, "activity": worker_activity(w)})
|
|
58
|
+
except Exception as exc: # noqa: BLE001
|
|
59
|
+
return jsonify({"ok": False, "error": f"{type(exc).__name__}: {exc}", "activity": []}), 503
|
|
60
|
+
|
|
61
|
+
@bp.get("/llm/workers/<worker_id>/health")
|
|
62
|
+
def workers_health(worker_id):
|
|
63
|
+
w = registry.lookup_worker(worker_id)
|
|
64
|
+
if w is None:
|
|
65
|
+
return jsonify({"reachable": False, "error": f"no worker {worker_id!r}"}), 404
|
|
66
|
+
url = (w.get("url") or "").rstrip("/") + "/health"
|
|
67
|
+
try:
|
|
68
|
+
with urllib.request.urlopen(url, timeout=4) as r:
|
|
69
|
+
return jsonify({"reachable": True, "url": url, "health": json.loads(r.read() or b"{}")})
|
|
70
|
+
except Exception as exc: # noqa: BLE001
|
|
71
|
+
return jsonify({"reachable": False, "url": url, "error": f"{type(exc).__name__}: {exc}"})
|
|
72
|
+
|
|
73
|
+
@bp.get("/llm/models/status")
|
|
74
|
+
def models_status():
|
|
75
|
+
from .routes import catalog_rows
|
|
76
|
+
rows = catalog_rows()
|
|
77
|
+
models = [{"model_key": r["model_key"], "framework": r.get("framework"), "status": r.get("status") or "known",
|
|
78
|
+
"blocked": False, "archived": {"marked": False, "at": None, "by": None, "reason": None},
|
|
79
|
+
"admission": {"status": "unknown"}, "task": r.get("task")} for r in rows]
|
|
80
|
+
return jsonify({"models": models, "count": len(models), "buckets": {}, "counts": {},
|
|
81
|
+
"generated_at": time.time(), "live": True})
|
|
82
|
+
|
|
83
|
+
@bp.get("/llm/events")
|
|
84
|
+
def llm_events():
|
|
85
|
+
"""SSE: the full snapshot on connect, then a fresh one the moment the
|
|
86
|
+
registry changes (a worker registers or heartbeats). No polling."""
|
|
87
|
+
def gen():
|
|
88
|
+
seen = registry.version()
|
|
89
|
+
yield "event: snapshot\ndata: " + json.dumps({"ts": time.time(), "versions": {"workers": seen},
|
|
90
|
+
"feeds": snapshot_payload()}, default=str) + "\n\n"
|
|
91
|
+
started = time.time()
|
|
92
|
+
while time.time() - started < 3600:
|
|
93
|
+
v = registry.wait_change(seen, 15.0)
|
|
94
|
+
if v == seen:
|
|
95
|
+
yield ": keepalive\n\n"
|
|
96
|
+
continue
|
|
97
|
+
seen = v
|
|
98
|
+
feeds = snapshot_payload()
|
|
99
|
+
for name, payload in feeds.items():
|
|
100
|
+
yield "event: feed\ndata: " + json.dumps({"feed": name, "version": v, "updated": time.time(),
|
|
101
|
+
"payload": payload}, default=str) + "\n\n"
|
|
102
|
+
return Response(stream_with_context(gen()), mimetype="text/event-stream",
|
|
103
|
+
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|
|
104
|
+
|
|
105
|
+
@bp.get("/readiness")
|
|
106
|
+
def readiness():
|
|
107
|
+
"""The console's landing / health probe: hugpy_core is up; serving and
|
|
108
|
+
storage are read live (registry + the model drive)."""
|
|
109
|
+
import os
|
|
110
|
+
from .routes import catalog_rows
|
|
111
|
+
root = os.environ.get("HUGPY_CORE_STORAGE_ROOT", "/mnt/16T_toshiba/llm_storage")
|
|
112
|
+
st = os.statvfs(root) if os.path.isdir(root) else None
|
|
113
|
+
online = [w for w in registry.all_workers() if registry.is_online(w)]
|
|
114
|
+
serving = any((w.get("loaded_models") or []) for w in online)
|
|
115
|
+
return jsonify({
|
|
116
|
+
"central": {"auth_mode": "open", "version": "hugpy_core"},
|
|
117
|
+
"connect": {"base_url": request.host_url.rstrip("/")},
|
|
118
|
+
"console": {"configured": True, "url": None},
|
|
119
|
+
"serving": {"any_serving": serving, "enabled": bool(online), "slots": [],
|
|
120
|
+
"workers_online": len(online)},
|
|
121
|
+
"storage": {"exists": st is not None, "root": root,
|
|
122
|
+
"free_bytes": (st.f_bavail * st.f_frsize) if st else None,
|
|
123
|
+
"free_gb": round(st.f_bavail * st.f_frsize / 1e9, 1) if st else None,
|
|
124
|
+
"writable": os.access(root, os.W_OK), "model_count": len(catalog_rows())},
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
@bp.get("/llm/env-profiles")
|
|
128
|
+
def env_profiles():
|
|
129
|
+
return jsonify({"bases": ["worker"], "profiles": []})
|
|
130
|
+
|
|
131
|
+
@bp.get("/llm/model-groups")
|
|
132
|
+
def model_groups():
|
|
133
|
+
return jsonify({"groups": []})
|
|
134
|
+
|
|
135
|
+
@bp.get("/models/notes")
|
|
136
|
+
def models_notes():
|
|
137
|
+
return jsonify({"flags": NOTE_FLAGS, "notes": {}})
|
|
138
|
+
|
|
139
|
+
@bp.get("/llm/fleet/distribution")
|
|
140
|
+
def fleet_distribution():
|
|
141
|
+
return jsonify({"env_override": False, "mode": "feasible", "source": "default", "stored": None})
|
|
142
|
+
|
|
143
|
+
@bp.get("/llm/enroll-tokens")
|
|
144
|
+
def enroll_tokens():
|
|
145
|
+
return jsonify([])
|
|
146
|
+
|
|
147
|
+
@bp.get("/llm/help/tickets")
|
|
148
|
+
def help_tickets():
|
|
149
|
+
return jsonify({"tickets": [], "count": 0})
|
|
150
|
+
|
|
151
|
+
@bp.post("/llm/sessions/<session_id>/lease")
|
|
152
|
+
def session_lease(session_id):
|
|
153
|
+
body = request.get_json(silent=True) or {}
|
|
154
|
+
return jsonify({"ok": True, "session_id": session_id, "state": body.get("state") or "active",
|
|
155
|
+
"lease_fresh": True})
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# LIFTED from legacy hugpy (read-only reference):
|
|
2
|
+
# model_key_forms <- py/foundation/hugpy_platform/src/hugpy_platform/model_keys.py:8-22 (verbatim)
|
|
3
|
+
# canonical_key <- py/inference/hugpy_engine/src/hugpy_engine/fit/types.py:53-60 (verbatim)
|
|
4
|
+
# _canon_forms, serveable_match <- py/fleet/hugpy_fleet/src/hugpy_fleet/central/workers.py:3724,3762
|
|
5
|
+
# DROPPED: the blocked-sibling guard in serveable_match (no blocklist here).
|
|
6
|
+
"""Model-key identity: alias forms and the one membership predicate routing uses."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def model_key_forms(model_key: Any) -> set[str]:
|
|
14
|
+
"""Return the raw key, case variants, and its ``/`` and ``~`` tails."""
|
|
15
|
+
if not model_key:
|
|
16
|
+
return set()
|
|
17
|
+
raw = str(model_key).strip()
|
|
18
|
+
forms = {raw, raw.lower()}
|
|
19
|
+
tail = raw.split("/")[-1]
|
|
20
|
+
forms.add(tail)
|
|
21
|
+
forms.add(tail.lower())
|
|
22
|
+
if "~" in raw:
|
|
23
|
+
base = raw.split("~", 1)[1]
|
|
24
|
+
if base:
|
|
25
|
+
forms.add(base)
|
|
26
|
+
forms.add(base.lower())
|
|
27
|
+
return forms
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Case is not a name token (the routing alias union already folds it); the
|
|
31
|
+
# separator before the suffix may be ``-`` or ``_`` (``Qwen3.8_4B_Distilled_GGUF``
|
|
32
|
+
# is a live key) — nothing else is folded.
|
|
33
|
+
_FORMAT_SUFFIX_RE = re.compile(r"[-_]gguf$", re.IGNORECASE)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def canonical_key(model_key: Any) -> str:
|
|
37
|
+
"""``model_key`` with ONLY the trailing format suffix (``-GGUF``) removed.
|
|
38
|
+
Every other token is identity and is kept verbatim."""
|
|
39
|
+
s = str(model_key or "").strip()
|
|
40
|
+
return _FORMAT_SUFFIX_RE.sub("", s, count=1)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _canon_forms(forms) -> set:
|
|
44
|
+
"""The alias forms of a key reduced to their canonical identity (lower-cased,
|
|
45
|
+
``-GGUF`` stripped) — the set two keys must intersect on to be ONE model."""
|
|
46
|
+
return {canonical_key(f).lower() for f in (forms or ()) if f}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def serveable_match(model_key: str, wanted: set, serveable) -> bool:
|
|
50
|
+
"""True iff one of ``serveable``'s advertised keys names ``model_key``."""
|
|
51
|
+
want_c = _canon_forms(wanted)
|
|
52
|
+
for m in serveable:
|
|
53
|
+
if m == model_key:
|
|
54
|
+
return True
|
|
55
|
+
# Alias match = the "~"/"/"-tail union (k67) OR the ``-GGUF`` format
|
|
56
|
+
# equivalence (operator rule 2026-09-29). Both are IDENTITY: nothing
|
|
57
|
+
# here ever matches a key carrying an extra name token (``-Distill``,
|
|
58
|
+
# a quant marker…), so a worker holding ``X-Distill-GGUF`` is never
|
|
59
|
+
# credited with ``X-GGUF`` — not as resident, allocated, or on-disk.
|
|
60
|
+
if (wanted & model_key_forms(m)) or (want_c & _canon_forms(model_key_forms(m))):
|
|
61
|
+
return True
|
|
62
|
+
return False
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""LIVE window: what is loaded / loading / failed / free on each box, pushed the
|
|
2
|
+
instant it changes. inference_engine triggers notify 'live' (residents,
|
|
3
|
+
call_log) and 'gpu_state' (the card); one LISTEN thread re-reads and wakes every
|
|
4
|
+
open page. No polling anywhere."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import threading
|
|
9
|
+
import time
|
|
10
|
+
|
|
11
|
+
from flask import Blueprint, Response, stream_with_context
|
|
12
|
+
|
|
13
|
+
IE_DSN = "dbname=inference_engine"
|
|
14
|
+
_cond = threading.Condition()
|
|
15
|
+
_state = {"version": 0, "data": {}}
|
|
16
|
+
_started = [False]
|
|
17
|
+
_lock = threading.Lock()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _read(c) -> dict:
|
|
21
|
+
res = [dict(zip(("worker", "key", "state", "pid", "ctx", "ngl", "total_layers", "vram", "ram", "error",
|
|
22
|
+
"engine", "since", "last_used", "calls", "busy", "loading"), r))
|
|
23
|
+
for r in c.execute("SELECT worker, key, state, pid, ctx, n_gpu_layers, total_layers, vram_bytes, ram_bytes, "
|
|
24
|
+
"load_error, engine, extract(epoch from resident_since), extract(epoch from last_used), "
|
|
25
|
+
"calls, busy, loading FROM residents ORDER BY worker, key").fetchall()]
|
|
26
|
+
gpu = [dict(zip(("worker", "device", "name", "total", "usable", "used", "free", "reason", "at"), r))
|
|
27
|
+
for r in c.execute("SELECT worker, device, name, total, usable, used, free, reason, extract(epoch from at) "
|
|
28
|
+
"FROM gpu_state ORDER BY worker, device").fetchall()]
|
|
29
|
+
host = [dict(zip(("worker", "mem_total", "mem_available"), r))
|
|
30
|
+
for r in c.execute("SELECT worker, mem_total, mem_available FROM host_state").fetchall()]
|
|
31
|
+
calls = [dict(zip(("t", "worker", "model", "task", "status", "secs", "ctx", "error", "mode"), r))
|
|
32
|
+
for r in c.execute("SELECT extract(epoch from received_at), worker, model_name, task, http_status, "
|
|
33
|
+
"extract(epoch from (coalesce(responded_at, now()) - received_at)), context_length, "
|
|
34
|
+
"left(error, 160), response_meta->'decision'->>'mode' "
|
|
35
|
+
"FROM call_log WHERE http_status IS NOT NULL OR dispatched_at IS NOT NULL "
|
|
36
|
+
"ORDER BY received_at DESC LIMIT 25").fetchall()] # front-door rows (central-only stubs excluded)
|
|
37
|
+
return {"residents": res, "gpu": gpu, "host": host, "calls": calls, "at": time.time()}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _loop():
|
|
41
|
+
import psycopg
|
|
42
|
+
while True:
|
|
43
|
+
try:
|
|
44
|
+
with psycopg.connect(IE_DSN, autocommit=True) as conn, psycopg.connect(IE_DSN, autocommit=True) as rc:
|
|
45
|
+
conn.execute("LISTEN live")
|
|
46
|
+
conn.execute("LISTEN gpu_state")
|
|
47
|
+
while True:
|
|
48
|
+
d = _read(rc)
|
|
49
|
+
with _cond:
|
|
50
|
+
_state["data"] = d
|
|
51
|
+
_state["version"] += 1
|
|
52
|
+
_cond.notify_all()
|
|
53
|
+
list(conn.notifies(stop_after=1))
|
|
54
|
+
list(conn.notifies(timeout=0)) # coalesce a burst into one re-read
|
|
55
|
+
except Exception as exc: # noqa: BLE001
|
|
56
|
+
print(f"[live] reconnect: {exc}")
|
|
57
|
+
time.sleep(2)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _ensure():
|
|
61
|
+
with _lock:
|
|
62
|
+
if not _started[0]:
|
|
63
|
+
_started[0] = True
|
|
64
|
+
threading.Thread(target=_loop, name="live-listen", daemon=True).start()
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
PAGE = r"""<!doctype html><html><head><meta charset="utf-8"><title>hugpy live</title>
|
|
68
|
+
<style>
|
|
69
|
+
body{font:13px/1.4 ui-monospace,Menlo,monospace;background:#0f1115;color:#d6d9e0;margin:16px}
|
|
70
|
+
h2{font-size:14px;margin:18px 0 6px;color:#9aa4b2}
|
|
71
|
+
table{border-collapse:collapse;width:100%}td,th{padding:3px 8px;border-bottom:1px solid #222733;text-align:left;white-space:nowrap}
|
|
72
|
+
th{color:#7d8696;font-weight:normal}.n{text-align:right}
|
|
73
|
+
.loading{color:#f5b942}.serving{color:#5aa9ff}.idle{color:#4cc38a}.error,.dead{color:#ff6b6b}.fail{color:#ff6b6b}.ok{color:#4cc38a}
|
|
74
|
+
.bar{height:16px;background:#222733;position:relative;margin:4px 0}.bar i{position:absolute;top:0;bottom:0;left:0;background:#5aa9ff}
|
|
75
|
+
#st{color:#7d8696}
|
|
76
|
+
</style></head><body>
|
|
77
|
+
<div id=st>connecting…</div>
|
|
78
|
+
<h2>GPU</h2><div id=gpu></div>
|
|
79
|
+
<h2>Models on the boxes (fitevict front door)</h2><table id=res></table>
|
|
80
|
+
<h2>Recent calls</h2><table id=calls></table>
|
|
81
|
+
<script>
|
|
82
|
+
const B=n=>n==null?'—':Number(n).toLocaleString('en-US')+' B';
|
|
83
|
+
const T=e=>e?new Date(e*1000).toLocaleTimeString():'—';
|
|
84
|
+
function draw(d){
|
|
85
|
+
document.getElementById('st').textContent='live · updated '+T(d.at);
|
|
86
|
+
document.getElementById('gpu').innerHTML=d.gpu.map(g=>{
|
|
87
|
+
const h=(d.host||[]).find(x=>x.worker===g.worker)||{};
|
|
88
|
+
const pct=g.usable?Math.min(100,100*g.used/g.usable):0;
|
|
89
|
+
return `<div>${g.worker} · ${g.name} · used ${B(g.used)} · free ${B(g.free)} of usable ${B(g.usable)} · RAM available ${B(h.mem_available)} · last change: ${g.reason} ${T(g.at)}</div><div class=bar><i style="width:${pct}%"></i></div>`}).join('');
|
|
90
|
+
document.getElementById('res').innerHTML='<tr><th>box</th><th>model</th><th>state</th><th>engine</th><th class=n>ctx</th><th class=n>gpu layers</th><th class=n>VRAM</th><th class=n>RAM</th><th>since</th><th class=n>calls</th><th>error</th></tr>'+
|
|
91
|
+
(d.residents.length?d.residents.map(r=>{const st=r.busy?'serving':(r.loading?'loading':r.state);
|
|
92
|
+
return `<tr><td>${r.worker}</td><td>${r.key}</td><td class="${st}">● ${st}</td><td>${r.engine||''}</td><td class=n>${r.ctx??'—'}</td><td class=n>${r.ngl==null?'—':(r.ngl<0?'all':r.ngl)}${r.total_layers?'/'+r.total_layers:''}</td><td class=n>${B(r.vram)}</td><td class=n>${B(r.ram)}</td><td>${T(r.since)}</td><td class=n>${r.calls??0}</td><td class=error>${r.error||''}</td></tr>`}).join(''):'<tr><td colspan=11>nothing loaded</td></tr>');
|
|
93
|
+
document.getElementById('calls').innerHTML='<tr><th>time</th><th>model</th><th>task</th><th>status</th><th class=n>secs</th><th class=n>ctx</th><th>mode</th><th>error</th></tr>'+
|
|
94
|
+
d.calls.map(c=>{const ok=c.status==200, run=c.status==null;
|
|
95
|
+
return `<tr><td>${T(c.t)}</td><td>${c.model}</td><td>${c.task||''}</td><td class="${run?'loading':(ok?'ok':'fail')}">${run?'running':c.status}</td><td class=n>${c.secs==null?'':Number(c.secs).toFixed(1)}</td><td class=n>${c.ctx??''}</td><td>${c.mode||''}</td><td class=fail>${c.error||''}</td></tr>`}).join('');
|
|
96
|
+
}
|
|
97
|
+
function go(){const es=new EventSource('live/stream');es.onmessage=e=>draw(JSON.parse(e.data));
|
|
98
|
+
es.onerror=()=>{document.getElementById('st').textContent='reconnecting…'}}
|
|
99
|
+
go();
|
|
100
|
+
</script></body></html>"""
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def attach_live(bp: Blueprint) -> None:
|
|
104
|
+
@bp.get("/live")
|
|
105
|
+
def live_page():
|
|
106
|
+
_ensure()
|
|
107
|
+
return Response(PAGE, mimetype="text/html")
|
|
108
|
+
|
|
109
|
+
@bp.get("/live/stream")
|
|
110
|
+
def live_stream():
|
|
111
|
+
_ensure()
|
|
112
|
+
|
|
113
|
+
def gen():
|
|
114
|
+
seen = -1
|
|
115
|
+
started = time.time()
|
|
116
|
+
while time.time() - started < 3600:
|
|
117
|
+
with _cond:
|
|
118
|
+
if _state["version"] == seen:
|
|
119
|
+
_cond.wait(timeout=15)
|
|
120
|
+
v, d = _state["version"], _state["data"]
|
|
121
|
+
if v == seen:
|
|
122
|
+
yield ": keepalive\n\n"
|
|
123
|
+
continue
|
|
124
|
+
seen = v
|
|
125
|
+
yield "data: " + json.dumps(d, default=str) + "\n\n"
|
|
126
|
+
return Response(stream_with_context(gen()), mimetype="text/event-stream",
|
|
127
|
+
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|