cloxy 5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cloxy/__init__.py +3 -0
- cloxy/__main__.py +3 -0
- cloxy/backends/__init__.py +0 -0
- cloxy/backends/mlx_backend.py +228 -0
- cloxy/catalog.py +93 -0
- cloxy/cli.py +429 -0
- cloxy/config.py +106 -0
- cloxy/convos.py +421 -0
- cloxy/hardware.py +79 -0
- cloxy/mcp_server.py +199 -0
- cloxy/memory.py +752 -0
- cloxy/proxy.py +190 -0
- cloxy/server.py +557 -0
- cloxy-5.1.dist-info/METADATA +281 -0
- cloxy-5.1.dist-info/RECORD +19 -0
- cloxy-5.1.dist-info/WHEEL +5 -0
- cloxy-5.1.dist-info/entry_points.txt +2 -0
- cloxy-5.1.dist-info/licenses/LICENSE +21 -0
- cloxy-5.1.dist-info/top_level.txt +1 -0
cloxy/__init__.py
ADDED
cloxy/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""
|
|
2
|
+
MLX backend for Cloxy.
|
|
3
|
+
|
|
4
|
+
Wraps mlx-lm to provide a small async-friendly interface that the FastAPI
|
|
5
|
+
layer can call. Exposes:
|
|
6
|
+
- load_model(hf_id) — pulls the model into unified memory
|
|
7
|
+
- generate_chat(messages, max_tokens, ...) — non-streaming
|
|
8
|
+
- stream_chat(messages, max_tokens, ...) — async generator of StreamPiece
|
|
9
|
+
- is_loaded() / current_model() — introspection
|
|
10
|
+
|
|
11
|
+
Generation is serialized through one lock: MLX runs a single model on a single
|
|
12
|
+
Metal command queue, so concurrent generate() calls contend on the GPU and
|
|
13
|
+
balloon memory instead of running in parallel.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import asyncio
|
|
18
|
+
import threading
|
|
19
|
+
import time
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from typing import AsyncGenerator, List, Optional
|
|
22
|
+
|
|
23
|
+
# mlx-lm imports are deferred until load_model() so that
|
|
24
|
+
# `import backends.mlx_backend` is cheap and doesn't pull MLX
|
|
25
|
+
# into memory at FastAPI startup unless an LLM is configured.
|
|
26
|
+
|
|
27
|
+
_model = None
|
|
28
|
+
_tokenizer = None
|
|
29
|
+
_current_hf_id: Optional[str] = None
|
|
30
|
+
_load_lock = asyncio.Lock()
|
|
31
|
+
_gen_lock = asyncio.Lock() # one generation at a time on the GPU
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class StreamPiece:
|
|
36
|
+
"""One streamed fragment. finish_reason/usage are set only on the last piece."""
|
|
37
|
+
text: str
|
|
38
|
+
finish_reason: Optional[str] = None
|
|
39
|
+
usage: Optional[dict] = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def is_loaded() -> bool:
|
|
43
|
+
return _model is not None and _tokenizer is not None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def current_model() -> Optional[str]:
|
|
47
|
+
return _current_hf_id
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
async def load_model(hf_id: str) -> None:
|
|
51
|
+
"""
|
|
52
|
+
Pull `hf_id` from Hugging Face (or cache) and load into unified memory.
|
|
53
|
+
Idempotent — calling again with the same id is a no-op.
|
|
54
|
+
"""
|
|
55
|
+
global _model, _tokenizer, _current_hf_id
|
|
56
|
+
|
|
57
|
+
if _current_hf_id == hf_id and is_loaded():
|
|
58
|
+
return
|
|
59
|
+
|
|
60
|
+
async with _load_lock:
|
|
61
|
+
if _current_hf_id == hf_id and is_loaded():
|
|
62
|
+
return
|
|
63
|
+
|
|
64
|
+
# Defer import so module import is cheap.
|
|
65
|
+
from mlx_lm import load
|
|
66
|
+
|
|
67
|
+
# mlx_lm.load is blocking; run it off the event loop.
|
|
68
|
+
loop = asyncio.get_running_loop()
|
|
69
|
+
model, tokenizer = await loop.run_in_executor(None, load, hf_id)
|
|
70
|
+
|
|
71
|
+
_model = model
|
|
72
|
+
_tokenizer = tokenizer
|
|
73
|
+
_current_hf_id = hf_id
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _apply_chat_template(messages: List[dict]) -> str:
|
|
77
|
+
"""
|
|
78
|
+
Render a list of {role, content} messages into a model-specific prompt
|
|
79
|
+
string using the tokenizer's chat template.
|
|
80
|
+
"""
|
|
81
|
+
if _tokenizer is None:
|
|
82
|
+
raise RuntimeError("Model not loaded. Call load_model() first.")
|
|
83
|
+
return _tokenizer.apply_chat_template(
|
|
84
|
+
messages,
|
|
85
|
+
tokenize=False,
|
|
86
|
+
add_generation_prompt=True,
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _usage(prompt_tokens: int, completion_tokens: int) -> dict:
|
|
91
|
+
return {
|
|
92
|
+
"prompt_tokens": prompt_tokens,
|
|
93
|
+
"completion_tokens": completion_tokens,
|
|
94
|
+
"total_tokens": prompt_tokens + completion_tokens,
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
async def generate_chat(
|
|
99
|
+
messages: List[dict],
|
|
100
|
+
max_tokens: int = 512,
|
|
101
|
+
temperature: float = 0.7,
|
|
102
|
+
top_p: float = 0.95,
|
|
103
|
+
) -> dict:
|
|
104
|
+
"""
|
|
105
|
+
Non-streaming chat completion. Returns an OpenAI-compatible dict.
|
|
106
|
+
"""
|
|
107
|
+
if not is_loaded():
|
|
108
|
+
raise RuntimeError("Model not loaded.")
|
|
109
|
+
|
|
110
|
+
from mlx_lm import generate
|
|
111
|
+
from mlx_lm.sample_utils import make_sampler
|
|
112
|
+
|
|
113
|
+
prompt = _apply_chat_template(messages)
|
|
114
|
+
sampler = make_sampler(temp=temperature, top_p=top_p)
|
|
115
|
+
|
|
116
|
+
loop = asyncio.get_running_loop()
|
|
117
|
+
async with _gen_lock:
|
|
118
|
+
started = time.perf_counter()
|
|
119
|
+
text = await loop.run_in_executor(
|
|
120
|
+
None,
|
|
121
|
+
lambda: generate(
|
|
122
|
+
_model,
|
|
123
|
+
_tokenizer,
|
|
124
|
+
prompt=prompt,
|
|
125
|
+
max_tokens=max_tokens,
|
|
126
|
+
sampler=sampler,
|
|
127
|
+
verbose=False,
|
|
128
|
+
),
|
|
129
|
+
)
|
|
130
|
+
elapsed = time.perf_counter() - started
|
|
131
|
+
|
|
132
|
+
# mlx-lm returns the generated text only (not the prompt).
|
|
133
|
+
completion_tokens = len(_tokenizer.encode(text))
|
|
134
|
+
prompt_tokens = len(_tokenizer.encode(prompt))
|
|
135
|
+
# generate() doesn't report why it stopped; hitting the budget is the tell.
|
|
136
|
+
finish_reason = "length" if completion_tokens >= max_tokens else "stop"
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
"id": f"cloxy-{int(time.time() * 1000)}",
|
|
140
|
+
"object": "chat.completion",
|
|
141
|
+
"created": int(time.time()),
|
|
142
|
+
"model": _current_hf_id,
|
|
143
|
+
"choices": [{
|
|
144
|
+
"index": 0,
|
|
145
|
+
"message": {"role": "assistant", "content": text},
|
|
146
|
+
"finish_reason": finish_reason,
|
|
147
|
+
}],
|
|
148
|
+
"usage": _usage(prompt_tokens, completion_tokens),
|
|
149
|
+
"cloxy_meta": {
|
|
150
|
+
"elapsed_seconds": round(elapsed, 3),
|
|
151
|
+
"tokens_per_second": round(completion_tokens / elapsed, 2) if elapsed > 0 else None,
|
|
152
|
+
},
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
async def stream_chat(
|
|
157
|
+
messages: List[dict],
|
|
158
|
+
max_tokens: int = 512,
|
|
159
|
+
temperature: float = 0.7,
|
|
160
|
+
top_p: float = 0.95,
|
|
161
|
+
) -> AsyncGenerator[StreamPiece, None]:
|
|
162
|
+
"""
|
|
163
|
+
Streaming chat completion. Yields StreamPiece fragments as they are
|
|
164
|
+
produced; the last piece carries finish_reason ("stop" / "length") and
|
|
165
|
+
usage. The FastAPI route is responsible for SSE-framing them.
|
|
166
|
+
|
|
167
|
+
Closing the generator early (client disconnect) sets a stop flag that the
|
|
168
|
+
producer thread checks every token, so generation doesn't run on to
|
|
169
|
+
max_tokens for nobody.
|
|
170
|
+
"""
|
|
171
|
+
if not is_loaded():
|
|
172
|
+
raise RuntimeError("Model not loaded.")
|
|
173
|
+
|
|
174
|
+
from mlx_lm import stream_generate
|
|
175
|
+
from mlx_lm.sample_utils import make_sampler
|
|
176
|
+
|
|
177
|
+
prompt = _apply_chat_template(messages)
|
|
178
|
+
sampler = make_sampler(temp=temperature, top_p=top_p)
|
|
179
|
+
|
|
180
|
+
# stream_generate is a sync generator; bridge it to async via a thread.
|
|
181
|
+
loop = asyncio.get_running_loop()
|
|
182
|
+
queue: asyncio.Queue = asyncio.Queue()
|
|
183
|
+
stop = threading.Event()
|
|
184
|
+
_SENTINEL = object()
|
|
185
|
+
|
|
186
|
+
def _producer():
|
|
187
|
+
try:
|
|
188
|
+
last = None
|
|
189
|
+
for response in stream_generate(
|
|
190
|
+
_model,
|
|
191
|
+
_tokenizer,
|
|
192
|
+
prompt=prompt,
|
|
193
|
+
max_tokens=max_tokens,
|
|
194
|
+
sampler=sampler,
|
|
195
|
+
):
|
|
196
|
+
if stop.is_set():
|
|
197
|
+
break
|
|
198
|
+
last = response
|
|
199
|
+
# Newer mlx-lm yields a GenerationResponse with .text;
|
|
200
|
+
# older versions yield a raw string.
|
|
201
|
+
text = getattr(response, "text", response)
|
|
202
|
+
loop.call_soon_threadsafe(queue.put_nowait, StreamPiece(text=text))
|
|
203
|
+
if last is not None and not stop.is_set():
|
|
204
|
+
finish = getattr(last, "finish_reason", None) or "stop"
|
|
205
|
+
p_tok = getattr(last, "prompt_tokens", None)
|
|
206
|
+
g_tok = getattr(last, "generation_tokens", None)
|
|
207
|
+
usage = _usage(p_tok, g_tok) if p_tok is not None and g_tok is not None else None
|
|
208
|
+
loop.call_soon_threadsafe(
|
|
209
|
+
queue.put_nowait, StreamPiece(text="", finish_reason=finish, usage=usage)
|
|
210
|
+
)
|
|
211
|
+
except Exception as e: # surface generation errors to the consumer
|
|
212
|
+
loop.call_soon_threadsafe(queue.put_nowait, e)
|
|
213
|
+
finally:
|
|
214
|
+
loop.call_soon_threadsafe(queue.put_nowait, _SENTINEL)
|
|
215
|
+
|
|
216
|
+
async with _gen_lock:
|
|
217
|
+
fut = loop.run_in_executor(None, _producer)
|
|
218
|
+
try:
|
|
219
|
+
while True:
|
|
220
|
+
item = await queue.get()
|
|
221
|
+
if item is _SENTINEL:
|
|
222
|
+
break
|
|
223
|
+
if isinstance(item, Exception):
|
|
224
|
+
raise item
|
|
225
|
+
yield item
|
|
226
|
+
finally:
|
|
227
|
+
stop.set()
|
|
228
|
+
await fut
|
cloxy/catalog.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Model catalog for Cloxy — MLX-Community quantized models.
|
|
3
|
+
|
|
4
|
+
Each entry is a real model on Hugging Face that mlx-lm can load directly.
|
|
5
|
+
`load_gb` is the conservative memory footprint after load (weights + KV cache
|
|
6
|
+
for a reasonable context window). Recommendations leave headroom inside that
|
|
7
|
+
already-conservative number.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import List
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Model:
|
|
17
|
+
short_name: str
|
|
18
|
+
hf_id: str
|
|
19
|
+
load_gb: float
|
|
20
|
+
tier: str
|
|
21
|
+
blurb: str
|
|
22
|
+
|
|
23
|
+
def line(self) -> str:
|
|
24
|
+
return f"{self.short_name:<22} {self.load_gb:>5.1f} GB {self.tier:<8} {self.blurb}"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# Curated MLX-Community models. Keep this list short and trusted.
|
|
28
|
+
# Add more later as they prove out.
|
|
29
|
+
CATALOG: List[Model] = [
|
|
30
|
+
Model("Phi-3 mini 4B",
|
|
31
|
+
"mlx-community/Phi-3-mini-4k-instruct-4bit",
|
|
32
|
+
2.5, "tiny",
|
|
33
|
+
"Fastest, lowest RAM. Good for chat, weak at reasoning."),
|
|
34
|
+
Model("Qwen 2.5 7B",
|
|
35
|
+
"mlx-community/Qwen2.5-7B-Instruct-4bit",
|
|
36
|
+
5.0, "small",
|
|
37
|
+
"Strong all-rounder for its size."),
|
|
38
|
+
Model("Llama 3.1 8B",
|
|
39
|
+
"mlx-community/Meta-Llama-3.1-8B-Instruct-4bit",
|
|
40
|
+
5.5, "small",
|
|
41
|
+
"Meta's workhorse. Broad knowledge."),
|
|
42
|
+
Model("Qwen 2.5 14B",
|
|
43
|
+
"mlx-community/Qwen2.5-14B-Instruct-4bit",
|
|
44
|
+
9.0, "medium",
|
|
45
|
+
"Sweet spot for 32 GB Macs."),
|
|
46
|
+
Model("Qwen 2.5 32B",
|
|
47
|
+
"mlx-community/Qwen2.5-32B-Instruct-4bit",
|
|
48
|
+
19.0, "large",
|
|
49
|
+
"Excellent reasoning. Best pick for 64 GB Macs."),
|
|
50
|
+
Model("Llama 3.1 70B",
|
|
51
|
+
"mlx-community/Meta-Llama-3.1-70B-Instruct-4bit",
|
|
52
|
+
40.0, "xlarge",
|
|
53
|
+
"Top-tier capability. Needs a Studio or M4 Max."),
|
|
54
|
+
Model("Qwen 2.5 72B",
|
|
55
|
+
"mlx-community/Qwen2.5-72B-Instruct-4bit",
|
|
56
|
+
41.0, "xlarge",
|
|
57
|
+
"Very strong, especially on code + math."),
|
|
58
|
+
Model("Llama 3.1 405B",
|
|
59
|
+
"mlx-community/Meta-Llama-3.1-405B-Instruct-4bit",
|
|
60
|
+
220.0, "absurd",
|
|
61
|
+
"Yes, this exists. Yes, your Mac probably can't run it."),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def fits(model: Model, available_gb: float, context_headroom: float = 1.2) -> bool:
|
|
66
|
+
"""
|
|
67
|
+
A model 'fits' if its load_gb plus a context-window buffer is under the
|
|
68
|
+
AI-available memory budget. The 1.2x default leaves room for long contexts
|
|
69
|
+
and the system to breathe.
|
|
70
|
+
"""
|
|
71
|
+
return model.load_gb * context_headroom <= available_gb
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def recommend(available_gb: float, max_results: int = 5) -> List[Model]:
|
|
75
|
+
"""Return the largest-fitting models first, capped at max_results."""
|
|
76
|
+
fitting = [m for m in CATALOG if fits(m, available_gb)]
|
|
77
|
+
fitting.sort(key=lambda m: m.load_gb, reverse=True)
|
|
78
|
+
return fitting[:max_results]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def by_short_name(name: str) -> Model:
|
|
82
|
+
name_l = name.strip().lower()
|
|
83
|
+
for m in CATALOG:
|
|
84
|
+
if m.short_name.lower() == name_l or m.hf_id.lower() == name_l:
|
|
85
|
+
return m
|
|
86
|
+
raise KeyError(f"No model matching '{name}'. Run `cloxy init` to see options.")
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
if __name__ == "__main__":
|
|
90
|
+
# Quick demo: pretend we have 64 GB
|
|
91
|
+
print("Demo: 64 GB unified memory, ~46 GB available for AI\n")
|
|
92
|
+
for m in recommend(available_gb=46.0):
|
|
93
|
+
print(" " + m.line())
|