hugpy-core 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hugpy_core/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """hugpy_core — hugpy central, minimal: live worker registry, model->worker
2
+ pick, and the OpenAI /v1 relay to workers (hugpy_worker over fitevict)."""
3
+ __version__ = "0.1.0"
hugpy_core/app.py ADDED
@@ -0,0 +1,30 @@
1
+ """hugpy_core router app: abstract_flask app over the routes module."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import logging
6
+
7
+ from abstract_flask import get_Flask_app
8
+
9
+ from . import routes
10
+
11
+ logger = logging.getLogger("hugpy_core")
12
+
13
+
14
+ def build_app():
15
+ return get_Flask_app(name="hugpy_core", routes=routes)
16
+
17
+
18
+ def main(argv=None) -> int:
19
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
20
+ ap = argparse.ArgumentParser(prog="hugpy_core")
21
+ ap.add_argument("--host", default="127.0.0.1")
22
+ ap.add_argument("--port", type=int, default=7020)
23
+ a = ap.parse_args(argv)
24
+ logger.info("hugpy_core router on %s:%s", a.host, a.port)
25
+ build_app().run(host=a.host, port=a.port, threaded=True)
26
+ return 0
27
+
28
+
29
+ if __name__ == "__main__":
30
+ raise SystemExit(main())
@@ -0,0 +1,155 @@
1
+ """The console's remaining routes, served by hugpy_core itself (no fallback to
2
+ the hugpy_next central). Live values come from the DB / the registry at the
3
+ moment of the call; /llm/events pushes on every registry change."""
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import time
8
+ import urllib.request
9
+
10
+ from flask import Blueprint, Response, jsonify, request, stream_with_context
11
+
12
+ from . import registry
13
+
14
+ IE_DSN = "dbname=inference_engine"
15
+ NOTE_FLAGS = ["broken", "trash", "archive", "unknown", "needs-env", "experimental", "keep"]
16
+
17
+
18
+ def _ie(sql: str, params: tuple = ()) -> list:
19
+ import psycopg
20
+ with psycopg.connect(IE_DSN, autocommit=True, connect_timeout=5) as c:
21
+ return c.execute(sql, params).fetchall()
22
+
23
+
24
+ def worker_activity(worker: dict) -> list:
25
+ """One host view of the worker's GPU: every process on its card from
26
+ inference_engine.gpu_state (published by its fitevict front door the moment
27
+ the card changes), joined to the front door's residents."""
28
+ name = worker.get("name")
29
+ cards = _ie("SELECT device, processes, hugpy_pids, probe_pids FROM gpu_state WHERE worker = %s", (name,))
30
+ res = {int(pid): key for key, pid in _ie("SELECT key, pid FROM residents WHERE worker = %s AND pid IS NOT NULL", (name,))}
31
+ rows = []
32
+ for device, procs, ours, probes in cards:
33
+ for pid_s, nbytes in sorted((procs or {}).items()):
34
+ pid = int(pid_s)
35
+ key = res.get(pid)
36
+ kind = "hugpy" if key or pid in (ours or []) else ("probe" if pid in (probes or []) else "observed")
37
+ rows.append({"kind": kind, "model_key": key, "service": "fitevict" if kind == "hugpy" else
38
+ ("llama.cpp dry run" if kind == "probe" else "gpu process"),
39
+ "pids": [pid], "gpu_indices": [device], "vram_bytes": int(nbytes),
40
+ "vram_mib": int(nbytes) // 1048576, "state": "running",
41
+ "evictable": kind == "hugpy", "immutable": kind == "observed",
42
+ "api_available": bool(key), "activity": {},
43
+ "note": ("Hugpy managed (fitevict)" if kind == "hugpy" else
44
+ "transient sizing probe" if kind == "probe" else
45
+ "GPU process observed; not managed by hugpy")})
46
+ return rows
47
+
48
+
49
+ def attach_console(bp: Blueprint, snapshot_payload) -> None:
50
+
51
+ @bp.get("/llm/workers/<worker_id>/activity")
52
+ def workers_activity(worker_id):
53
+ w = registry.lookup_worker(worker_id)
54
+ if w is None:
55
+ return jsonify({"ok": False, "error": f"no worker {worker_id!r}", "activity": []}), 404
56
+ try:
57
+ return jsonify({"ok": True, "activity": worker_activity(w)})
58
+ except Exception as exc: # noqa: BLE001
59
+ return jsonify({"ok": False, "error": f"{type(exc).__name__}: {exc}", "activity": []}), 503
60
+
61
+ @bp.get("/llm/workers/<worker_id>/health")
62
+ def workers_health(worker_id):
63
+ w = registry.lookup_worker(worker_id)
64
+ if w is None:
65
+ return jsonify({"reachable": False, "error": f"no worker {worker_id!r}"}), 404
66
+ url = (w.get("url") or "").rstrip("/") + "/health"
67
+ try:
68
+ with urllib.request.urlopen(url, timeout=4) as r:
69
+ return jsonify({"reachable": True, "url": url, "health": json.loads(r.read() or b"{}")})
70
+ except Exception as exc: # noqa: BLE001
71
+ return jsonify({"reachable": False, "url": url, "error": f"{type(exc).__name__}: {exc}"})
72
+
73
+ @bp.get("/llm/models/status")
74
+ def models_status():
75
+ from .routes import catalog_rows
76
+ rows = catalog_rows()
77
+ models = [{"model_key": r["model_key"], "framework": r.get("framework"), "status": r.get("status") or "known",
78
+ "blocked": False, "archived": {"marked": False, "at": None, "by": None, "reason": None},
79
+ "admission": {"status": "unknown"}, "task": r.get("task")} for r in rows]
80
+ return jsonify({"models": models, "count": len(models), "buckets": {}, "counts": {},
81
+ "generated_at": time.time(), "live": True})
82
+
83
+ @bp.get("/llm/events")
84
+ def llm_events():
85
+ """SSE: the full snapshot on connect, then a fresh one the moment the
86
+ registry changes (a worker registers or heartbeats). No polling."""
87
+ def gen():
88
+ seen = registry.version()
89
+ yield "event: snapshot\ndata: " + json.dumps({"ts": time.time(), "versions": {"workers": seen},
90
+ "feeds": snapshot_payload()}, default=str) + "\n\n"
91
+ started = time.time()
92
+ while time.time() - started < 3600:
93
+ v = registry.wait_change(seen, 15.0)
94
+ if v == seen:
95
+ yield ": keepalive\n\n"
96
+ continue
97
+ seen = v
98
+ feeds = snapshot_payload()
99
+ for name, payload in feeds.items():
100
+ yield "event: feed\ndata: " + json.dumps({"feed": name, "version": v, "updated": time.time(),
101
+ "payload": payload}, default=str) + "\n\n"
102
+ return Response(stream_with_context(gen()), mimetype="text/event-stream",
103
+ headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
104
+
105
+ @bp.get("/readiness")
106
+ def readiness():
107
+ """The console's landing / health probe: hugpy_core is up; serving and
108
+ storage are read live (registry + the model drive)."""
109
+ import os
110
+ from .routes import catalog_rows
111
+ root = os.environ.get("HUGPY_CORE_STORAGE_ROOT", "/mnt/16T_toshiba/llm_storage")
112
+ st = os.statvfs(root) if os.path.isdir(root) else None
113
+ online = [w for w in registry.all_workers() if registry.is_online(w)]
114
+ serving = any((w.get("loaded_models") or []) for w in online)
115
+ return jsonify({
116
+ "central": {"auth_mode": "open", "version": "hugpy_core"},
117
+ "connect": {"base_url": request.host_url.rstrip("/")},
118
+ "console": {"configured": True, "url": None},
119
+ "serving": {"any_serving": serving, "enabled": bool(online), "slots": [],
120
+ "workers_online": len(online)},
121
+ "storage": {"exists": st is not None, "root": root,
122
+ "free_bytes": (st.f_bavail * st.f_frsize) if st else None,
123
+ "free_gb": round(st.f_bavail * st.f_frsize / 1e9, 1) if st else None,
124
+ "writable": os.access(root, os.W_OK), "model_count": len(catalog_rows())},
125
+ })
126
+
127
+ @bp.get("/llm/env-profiles")
128
+ def env_profiles():
129
+ return jsonify({"bases": ["worker"], "profiles": []})
130
+
131
+ @bp.get("/llm/model-groups")
132
+ def model_groups():
133
+ return jsonify({"groups": []})
134
+
135
+ @bp.get("/models/notes")
136
+ def models_notes():
137
+ return jsonify({"flags": NOTE_FLAGS, "notes": {}})
138
+
139
+ @bp.get("/llm/fleet/distribution")
140
+ def fleet_distribution():
141
+ return jsonify({"env_override": False, "mode": "feasible", "source": "default", "stored": None})
142
+
143
+ @bp.get("/llm/enroll-tokens")
144
+ def enroll_tokens():
145
+ return jsonify([])
146
+
147
+ @bp.get("/llm/help/tickets")
148
+ def help_tickets():
149
+ return jsonify({"tickets": [], "count": 0})
150
+
151
+ @bp.post("/llm/sessions/<session_id>/lease")
152
+ def session_lease(session_id):
153
+ body = request.get_json(silent=True) or {}
154
+ return jsonify({"ok": True, "session_id": session_id, "state": body.get("state") or "active",
155
+ "lease_fresh": True})
hugpy_core/keys.py ADDED
@@ -0,0 +1,62 @@
1
+ # LIFTED from legacy hugpy (read-only reference):
2
+ # model_key_forms <- py/foundation/hugpy_platform/src/hugpy_platform/model_keys.py:8-22 (verbatim)
3
+ # canonical_key <- py/inference/hugpy_engine/src/hugpy_engine/fit/types.py:53-60 (verbatim)
4
+ # _canon_forms, serveable_match <- py/fleet/hugpy_fleet/src/hugpy_fleet/central/workers.py:3724,3762
5
+ # DROPPED: the blocked-sibling guard in serveable_match (no blocklist here).
6
+ """Model-key identity: alias forms and the one membership predicate routing uses."""
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ from typing import Any
11
+
12
+
13
+ def model_key_forms(model_key: Any) -> set[str]:
14
+ """Return the raw key, case variants, and its ``/`` and ``~`` tails."""
15
+ if not model_key:
16
+ return set()
17
+ raw = str(model_key).strip()
18
+ forms = {raw, raw.lower()}
19
+ tail = raw.split("/")[-1]
20
+ forms.add(tail)
21
+ forms.add(tail.lower())
22
+ if "~" in raw:
23
+ base = raw.split("~", 1)[1]
24
+ if base:
25
+ forms.add(base)
26
+ forms.add(base.lower())
27
+ return forms
28
+
29
+
30
+ # Case is not a name token (the routing alias union already folds it); the
31
+ # separator before the suffix may be ``-`` or ``_`` (``Qwen3.8_4B_Distilled_GGUF``
32
+ # is a live key) — nothing else is folded.
33
+ _FORMAT_SUFFIX_RE = re.compile(r"[-_]gguf$", re.IGNORECASE)
34
+
35
+
36
+ def canonical_key(model_key: Any) -> str:
37
+ """``model_key`` with ONLY the trailing format suffix (``-GGUF``) removed.
38
+ Every other token is identity and is kept verbatim."""
39
+ s = str(model_key or "").strip()
40
+ return _FORMAT_SUFFIX_RE.sub("", s, count=1)
41
+
42
+
43
+ def _canon_forms(forms) -> set:
44
+ """The alias forms of a key reduced to their canonical identity (lower-cased,
45
+ ``-GGUF`` stripped) — the set two keys must intersect on to be ONE model."""
46
+ return {canonical_key(f).lower() for f in (forms or ()) if f}
47
+
48
+
49
+ def serveable_match(model_key: str, wanted: set, serveable) -> bool:
50
+ """True iff one of ``serveable``'s advertised keys names ``model_key``."""
51
+ want_c = _canon_forms(wanted)
52
+ for m in serveable:
53
+ if m == model_key:
54
+ return True
55
+ # Alias match = the "~"/"/"-tail union (k67) OR the ``-GGUF`` format
56
+ # equivalence (operator rule 2026-09-29). Both are IDENTITY: nothing
57
+ # here ever matches a key carrying an extra name token (``-Distill``,
58
+ # a quant marker…), so a worker holding ``X-Distill-GGUF`` is never
59
+ # credited with ``X-GGUF`` — not as resident, allocated, or on-disk.
60
+ if (wanted & model_key_forms(m)) or (want_c & _canon_forms(model_key_forms(m))):
61
+ return True
62
+ return False
hugpy_core/live.py ADDED
@@ -0,0 +1,127 @@
1
+ """LIVE window: what is loaded / loading / failed / free on each box, pushed the
2
+ instant it changes. inference_engine triggers notify 'live' (residents,
3
+ call_log) and 'gpu_state' (the card); one LISTEN thread re-reads and wakes every
4
+ open page. No polling anywhere."""
5
+ from __future__ import annotations
6
+
7
+ import json
8
+ import threading
9
+ import time
10
+
11
+ from flask import Blueprint, Response, stream_with_context
12
+
13
+ IE_DSN = "dbname=inference_engine"
14
+ _cond = threading.Condition()
15
+ _state = {"version": 0, "data": {}}
16
+ _started = [False]
17
+ _lock = threading.Lock()
18
+
19
+
20
+ def _read(c) -> dict:
21
+ res = [dict(zip(("worker", "key", "state", "pid", "ctx", "ngl", "total_layers", "vram", "ram", "error",
22
+ "engine", "since", "last_used", "calls", "busy", "loading"), r))
23
+ for r in c.execute("SELECT worker, key, state, pid, ctx, n_gpu_layers, total_layers, vram_bytes, ram_bytes, "
24
+ "load_error, engine, extract(epoch from resident_since), extract(epoch from last_used), "
25
+ "calls, busy, loading FROM residents ORDER BY worker, key").fetchall()]
26
+ gpu = [dict(zip(("worker", "device", "name", "total", "usable", "used", "free", "reason", "at"), r))
27
+ for r in c.execute("SELECT worker, device, name, total, usable, used, free, reason, extract(epoch from at) "
28
+ "FROM gpu_state ORDER BY worker, device").fetchall()]
29
+ host = [dict(zip(("worker", "mem_total", "mem_available"), r))
30
+ for r in c.execute("SELECT worker, mem_total, mem_available FROM host_state").fetchall()]
31
+ calls = [dict(zip(("t", "worker", "model", "task", "status", "secs", "ctx", "error", "mode"), r))
32
+ for r in c.execute("SELECT extract(epoch from received_at), worker, model_name, task, http_status, "
33
+ "extract(epoch from (coalesce(responded_at, now()) - received_at)), context_length, "
34
+ "left(error, 160), response_meta->'decision'->>'mode' "
35
+ "FROM call_log WHERE http_status IS NOT NULL OR dispatched_at IS NOT NULL "
36
+ "ORDER BY received_at DESC LIMIT 25").fetchall()] # front-door rows (central-only stubs excluded)
37
+ return {"residents": res, "gpu": gpu, "host": host, "calls": calls, "at": time.time()}
38
+
39
+
40
+ def _loop():
41
+ import psycopg
42
+ while True:
43
+ try:
44
+ with psycopg.connect(IE_DSN, autocommit=True) as conn, psycopg.connect(IE_DSN, autocommit=True) as rc:
45
+ conn.execute("LISTEN live")
46
+ conn.execute("LISTEN gpu_state")
47
+ while True:
48
+ d = _read(rc)
49
+ with _cond:
50
+ _state["data"] = d
51
+ _state["version"] += 1
52
+ _cond.notify_all()
53
+ list(conn.notifies(stop_after=1))
54
+ list(conn.notifies(timeout=0)) # coalesce a burst into one re-read
55
+ except Exception as exc: # noqa: BLE001
56
+ print(f"[live] reconnect: {exc}")
57
+ time.sleep(2)
58
+
59
+
60
+ def _ensure():
61
+ with _lock:
62
+ if not _started[0]:
63
+ _started[0] = True
64
+ threading.Thread(target=_loop, name="live-listen", daemon=True).start()
65
+
66
+
67
+ PAGE = r"""<!doctype html><html><head><meta charset="utf-8"><title>hugpy live</title>
68
+ <style>
69
+ body{font:13px/1.4 ui-monospace,Menlo,monospace;background:#0f1115;color:#d6d9e0;margin:16px}
70
+ h2{font-size:14px;margin:18px 0 6px;color:#9aa4b2}
71
+ table{border-collapse:collapse;width:100%}td,th{padding:3px 8px;border-bottom:1px solid #222733;text-align:left;white-space:nowrap}
72
+ th{color:#7d8696;font-weight:normal}.n{text-align:right}
73
+ .loading{color:#f5b942}.serving{color:#5aa9ff}.idle{color:#4cc38a}.error,.dead{color:#ff6b6b}.fail{color:#ff6b6b}.ok{color:#4cc38a}
74
+ .bar{height:16px;background:#222733;position:relative;margin:4px 0}.bar i{position:absolute;top:0;bottom:0;left:0;background:#5aa9ff}
75
+ #st{color:#7d8696}
76
+ </style></head><body>
77
+ <div id=st>connecting…</div>
78
+ <h2>GPU</h2><div id=gpu></div>
79
+ <h2>Models on the boxes (fitevict front door)</h2><table id=res></table>
80
+ <h2>Recent calls</h2><table id=calls></table>
81
+ <script>
82
+ const B=n=>n==null?'—':Number(n).toLocaleString('en-US')+' B';
83
+ const T=e=>e?new Date(e*1000).toLocaleTimeString():'—';
84
+ function draw(d){
85
+ document.getElementById('st').textContent='live · updated '+T(d.at);
86
+ document.getElementById('gpu').innerHTML=d.gpu.map(g=>{
87
+ const h=(d.host||[]).find(x=>x.worker===g.worker)||{};
88
+ const pct=g.usable?Math.min(100,100*g.used/g.usable):0;
89
+ return `<div>${g.worker} · ${g.name} · used ${B(g.used)} · free ${B(g.free)} of usable ${B(g.usable)} · RAM available ${B(h.mem_available)} · last change: ${g.reason} ${T(g.at)}</div><div class=bar><i style="width:${pct}%"></i></div>`}).join('');
90
+ document.getElementById('res').innerHTML='<tr><th>box</th><th>model</th><th>state</th><th>engine</th><th class=n>ctx</th><th class=n>gpu layers</th><th class=n>VRAM</th><th class=n>RAM</th><th>since</th><th class=n>calls</th><th>error</th></tr>'+
91
+ (d.residents.length?d.residents.map(r=>{const st=r.busy?'serving':(r.loading?'loading':r.state);
92
+ return `<tr><td>${r.worker}</td><td>${r.key}</td><td class="${st}">● ${st}</td><td>${r.engine||''}</td><td class=n>${r.ctx??'—'}</td><td class=n>${r.ngl==null?'—':(r.ngl<0?'all':r.ngl)}${r.total_layers?'/'+r.total_layers:''}</td><td class=n>${B(r.vram)}</td><td class=n>${B(r.ram)}</td><td>${T(r.since)}</td><td class=n>${r.calls??0}</td><td class=error>${r.error||''}</td></tr>`}).join(''):'<tr><td colspan=11>nothing loaded</td></tr>');
93
+ document.getElementById('calls').innerHTML='<tr><th>time</th><th>model</th><th>task</th><th>status</th><th class=n>secs</th><th class=n>ctx</th><th>mode</th><th>error</th></tr>'+
94
+ d.calls.map(c=>{const ok=c.status==200, run=c.status==null;
95
+ return `<tr><td>${T(c.t)}</td><td>${c.model}</td><td>${c.task||''}</td><td class="${run?'loading':(ok?'ok':'fail')}">${run?'running':c.status}</td><td class=n>${c.secs==null?'':Number(c.secs).toFixed(1)}</td><td class=n>${c.ctx??''}</td><td>${c.mode||''}</td><td class=fail>${c.error||''}</td></tr>`}).join('');
96
+ }
97
+ function go(){const es=new EventSource('live/stream');es.onmessage=e=>draw(JSON.parse(e.data));
98
+ es.onerror=()=>{document.getElementById('st').textContent='reconnecting…'}}
99
+ go();
100
+ </script></body></html>"""
101
+
102
+
103
+ def attach_live(bp: Blueprint) -> None:
104
+ @bp.get("/live")
105
+ def live_page():
106
+ _ensure()
107
+ return Response(PAGE, mimetype="text/html")
108
+
109
+ @bp.get("/live/stream")
110
+ def live_stream():
111
+ _ensure()
112
+
113
+ def gen():
114
+ seen = -1
115
+ started = time.time()
116
+ while time.time() - started < 3600:
117
+ with _cond:
118
+ if _state["version"] == seen:
119
+ _cond.wait(timeout=15)
120
+ v, d = _state["version"], _state["data"]
121
+ if v == seen:
122
+ yield ": keepalive\n\n"
123
+ continue
124
+ seen = v
125
+ yield "data: " + json.dumps(d, default=str) + "\n\n"
126
+ return Response(stream_with_context(gen()), mimetype="text/event-stream",
127
+ headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})