cloxy 5.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cloxy/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """CLOXY — give your local AI eyes and memory."""
2
+
3
+ __version__ = "5.1"
cloxy/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
File without changes
@@ -0,0 +1,228 @@
1
+ """
2
+ MLX backend for Cloxy.
3
+
4
+ Wraps mlx-lm to provide a small async-friendly interface that the FastAPI
5
+ layer can call. Exposes:
6
+ - load_model(hf_id) — pulls the model into unified memory
7
+ - generate_chat(messages, max_tokens, ...) — non-streaming
8
+ - stream_chat(messages, max_tokens, ...) — async generator of StreamPiece
9
+ - is_loaded() / current_model() — introspection
10
+
11
+ Generation is serialized through one lock: MLX runs a single model on a single
12
+ Metal command queue, so concurrent generate() calls contend on the GPU and
13
+ balloon memory instead of running in parallel.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import asyncio
18
+ import threading
19
+ import time
20
+ from dataclasses import dataclass
21
+ from typing import AsyncGenerator, List, Optional
22
+
23
+ # mlx-lm imports are deferred until load_model() so that
24
+ # `import backends.mlx_backend` is cheap and doesn't pull MLX
25
+ # into memory at FastAPI startup unless an LLM is configured.
26
+
27
+ _model = None
28
+ _tokenizer = None
29
+ _current_hf_id: Optional[str] = None
30
+ _load_lock = asyncio.Lock()
31
+ _gen_lock = asyncio.Lock() # one generation at a time on the GPU
32
+
33
+
34
+ @dataclass
35
+ class StreamPiece:
36
+ """One streamed fragment. finish_reason/usage are set only on the last piece."""
37
+ text: str
38
+ finish_reason: Optional[str] = None
39
+ usage: Optional[dict] = None
40
+
41
+
42
+ def is_loaded() -> bool:
43
+ return _model is not None and _tokenizer is not None
44
+
45
+
46
+ def current_model() -> Optional[str]:
47
+ return _current_hf_id
48
+
49
+
50
+ async def load_model(hf_id: str) -> None:
51
+ """
52
+ Pull `hf_id` from Hugging Face (or cache) and load into unified memory.
53
+ Idempotent — calling again with the same id is a no-op.
54
+ """
55
+ global _model, _tokenizer, _current_hf_id
56
+
57
+ if _current_hf_id == hf_id and is_loaded():
58
+ return
59
+
60
+ async with _load_lock:
61
+ if _current_hf_id == hf_id and is_loaded():
62
+ return
63
+
64
+ # Defer import so module import is cheap.
65
+ from mlx_lm import load
66
+
67
+ # mlx_lm.load is blocking; run it off the event loop.
68
+ loop = asyncio.get_running_loop()
69
+ model, tokenizer = await loop.run_in_executor(None, load, hf_id)
70
+
71
+ _model = model
72
+ _tokenizer = tokenizer
73
+ _current_hf_id = hf_id
74
+
75
+
76
+ def _apply_chat_template(messages: List[dict]) -> str:
77
+ """
78
+ Render a list of {role, content} messages into a model-specific prompt
79
+ string using the tokenizer's chat template.
80
+ """
81
+ if _tokenizer is None:
82
+ raise RuntimeError("Model not loaded. Call load_model() first.")
83
+ return _tokenizer.apply_chat_template(
84
+ messages,
85
+ tokenize=False,
86
+ add_generation_prompt=True,
87
+ )
88
+
89
+
90
+ def _usage(prompt_tokens: int, completion_tokens: int) -> dict:
91
+ return {
92
+ "prompt_tokens": prompt_tokens,
93
+ "completion_tokens": completion_tokens,
94
+ "total_tokens": prompt_tokens + completion_tokens,
95
+ }
96
+
97
+
98
+ async def generate_chat(
99
+ messages: List[dict],
100
+ max_tokens: int = 512,
101
+ temperature: float = 0.7,
102
+ top_p: float = 0.95,
103
+ ) -> dict:
104
+ """
105
+ Non-streaming chat completion. Returns an OpenAI-compatible dict.
106
+ """
107
+ if not is_loaded():
108
+ raise RuntimeError("Model not loaded.")
109
+
110
+ from mlx_lm import generate
111
+ from mlx_lm.sample_utils import make_sampler
112
+
113
+ prompt = _apply_chat_template(messages)
114
+ sampler = make_sampler(temp=temperature, top_p=top_p)
115
+
116
+ loop = asyncio.get_running_loop()
117
+ async with _gen_lock:
118
+ started = time.perf_counter()
119
+ text = await loop.run_in_executor(
120
+ None,
121
+ lambda: generate(
122
+ _model,
123
+ _tokenizer,
124
+ prompt=prompt,
125
+ max_tokens=max_tokens,
126
+ sampler=sampler,
127
+ verbose=False,
128
+ ),
129
+ )
130
+ elapsed = time.perf_counter() - started
131
+
132
+ # mlx-lm returns the generated text only (not the prompt).
133
+ completion_tokens = len(_tokenizer.encode(text))
134
+ prompt_tokens = len(_tokenizer.encode(prompt))
135
+ # generate() doesn't report why it stopped; hitting the budget is the tell.
136
+ finish_reason = "length" if completion_tokens >= max_tokens else "stop"
137
+
138
+ return {
139
+ "id": f"cloxy-{int(time.time() * 1000)}",
140
+ "object": "chat.completion",
141
+ "created": int(time.time()),
142
+ "model": _current_hf_id,
143
+ "choices": [{
144
+ "index": 0,
145
+ "message": {"role": "assistant", "content": text},
146
+ "finish_reason": finish_reason,
147
+ }],
148
+ "usage": _usage(prompt_tokens, completion_tokens),
149
+ "cloxy_meta": {
150
+ "elapsed_seconds": round(elapsed, 3),
151
+ "tokens_per_second": round(completion_tokens / elapsed, 2) if elapsed > 0 else None,
152
+ },
153
+ }
154
+
155
+
156
+ async def stream_chat(
157
+ messages: List[dict],
158
+ max_tokens: int = 512,
159
+ temperature: float = 0.7,
160
+ top_p: float = 0.95,
161
+ ) -> AsyncGenerator[StreamPiece, None]:
162
+ """
163
+ Streaming chat completion. Yields StreamPiece fragments as they are
164
+ produced; the last piece carries finish_reason ("stop" / "length") and
165
+ usage. The FastAPI route is responsible for SSE-framing them.
166
+
167
+ Closing the generator early (client disconnect) sets a stop flag that the
168
+ producer thread checks every token, so generation doesn't run on to
169
+ max_tokens for nobody.
170
+ """
171
+ if not is_loaded():
172
+ raise RuntimeError("Model not loaded.")
173
+
174
+ from mlx_lm import stream_generate
175
+ from mlx_lm.sample_utils import make_sampler
176
+
177
+ prompt = _apply_chat_template(messages)
178
+ sampler = make_sampler(temp=temperature, top_p=top_p)
179
+
180
+ # stream_generate is a sync generator; bridge it to async via a thread.
181
+ loop = asyncio.get_running_loop()
182
+ queue: asyncio.Queue = asyncio.Queue()
183
+ stop = threading.Event()
184
+ _SENTINEL = object()
185
+
186
+ def _producer():
187
+ try:
188
+ last = None
189
+ for response in stream_generate(
190
+ _model,
191
+ _tokenizer,
192
+ prompt=prompt,
193
+ max_tokens=max_tokens,
194
+ sampler=sampler,
195
+ ):
196
+ if stop.is_set():
197
+ break
198
+ last = response
199
+ # Newer mlx-lm yields a GenerationResponse with .text;
200
+ # older versions yield a raw string.
201
+ text = getattr(response, "text", response)
202
+ loop.call_soon_threadsafe(queue.put_nowait, StreamPiece(text=text))
203
+ if last is not None and not stop.is_set():
204
+ finish = getattr(last, "finish_reason", None) or "stop"
205
+ p_tok = getattr(last, "prompt_tokens", None)
206
+ g_tok = getattr(last, "generation_tokens", None)
207
+ usage = _usage(p_tok, g_tok) if p_tok is not None and g_tok is not None else None
208
+ loop.call_soon_threadsafe(
209
+ queue.put_nowait, StreamPiece(text="", finish_reason=finish, usage=usage)
210
+ )
211
+ except Exception as e: # surface generation errors to the consumer
212
+ loop.call_soon_threadsafe(queue.put_nowait, e)
213
+ finally:
214
+ loop.call_soon_threadsafe(queue.put_nowait, _SENTINEL)
215
+
216
+ async with _gen_lock:
217
+ fut = loop.run_in_executor(None, _producer)
218
+ try:
219
+ while True:
220
+ item = await queue.get()
221
+ if item is _SENTINEL:
222
+ break
223
+ if isinstance(item, Exception):
224
+ raise item
225
+ yield item
226
+ finally:
227
+ stop.set()
228
+ await fut
cloxy/catalog.py ADDED
@@ -0,0 +1,93 @@
1
+ """
2
+ Model catalog for Cloxy — MLX-Community quantized models.
3
+
4
+ Each entry is a real model on Hugging Face that mlx-lm can load directly.
5
+ `load_gb` is the conservative memory footprint after load (weights + KV cache
6
+ for a reasonable context window). Recommendations leave headroom inside that
7
+ already-conservative number.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass
12
+ from typing import List
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class Model:
17
+ short_name: str
18
+ hf_id: str
19
+ load_gb: float
20
+ tier: str
21
+ blurb: str
22
+
23
+ def line(self) -> str:
24
+ return f"{self.short_name:<22} {self.load_gb:>5.1f} GB {self.tier:<8} {self.blurb}"
25
+
26
+
27
+ # Curated MLX-Community models. Keep this list short and trusted.
28
+ # Add more later as they prove out.
29
+ CATALOG: List[Model] = [
30
+ Model("Phi-3 mini 4B",
31
+ "mlx-community/Phi-3-mini-4k-instruct-4bit",
32
+ 2.5, "tiny",
33
+ "Fastest, lowest RAM. Good for chat, weak at reasoning."),
34
+ Model("Qwen 2.5 7B",
35
+ "mlx-community/Qwen2.5-7B-Instruct-4bit",
36
+ 5.0, "small",
37
+ "Strong all-rounder for its size."),
38
+ Model("Llama 3.1 8B",
39
+ "mlx-community/Meta-Llama-3.1-8B-Instruct-4bit",
40
+ 5.5, "small",
41
+ "Meta's workhorse. Broad knowledge."),
42
+ Model("Qwen 2.5 14B",
43
+ "mlx-community/Qwen2.5-14B-Instruct-4bit",
44
+ 9.0, "medium",
45
+ "Sweet spot for 32 GB Macs."),
46
+ Model("Qwen 2.5 32B",
47
+ "mlx-community/Qwen2.5-32B-Instruct-4bit",
48
+ 19.0, "large",
49
+ "Excellent reasoning. Best pick for 64 GB Macs."),
50
+ Model("Llama 3.1 70B",
51
+ "mlx-community/Meta-Llama-3.1-70B-Instruct-4bit",
52
+ 40.0, "xlarge",
53
+ "Top-tier capability. Needs a Studio or M4 Max."),
54
+ Model("Qwen 2.5 72B",
55
+ "mlx-community/Qwen2.5-72B-Instruct-4bit",
56
+ 41.0, "xlarge",
57
+ "Very strong, especially on code + math."),
58
+ Model("Llama 3.1 405B",
59
+ "mlx-community/Meta-Llama-3.1-405B-Instruct-4bit",
60
+ 220.0, "absurd",
61
+ "Yes, this exists. Yes, your Mac probably can't run it."),
62
+ ]
63
+
64
+
65
+ def fits(model: Model, available_gb: float, context_headroom: float = 1.2) -> bool:
66
+ """
67
+ A model 'fits' if its load_gb plus a context-window buffer is under the
68
+ AI-available memory budget. The 1.2x default leaves room for long contexts
69
+ and the system to breathe.
70
+ """
71
+ return model.load_gb * context_headroom <= available_gb
72
+
73
+
74
+ def recommend(available_gb: float, max_results: int = 5) -> List[Model]:
75
+ """Return the largest-fitting models first, capped at max_results."""
76
+ fitting = [m for m in CATALOG if fits(m, available_gb)]
77
+ fitting.sort(key=lambda m: m.load_gb, reverse=True)
78
+ return fitting[:max_results]
79
+
80
+
81
+ def by_short_name(name: str) -> Model:
82
+ name_l = name.strip().lower()
83
+ for m in CATALOG:
84
+ if m.short_name.lower() == name_l or m.hf_id.lower() == name_l:
85
+ return m
86
+ raise KeyError(f"No model matching '{name}'. Run `cloxy init` to see options.")
87
+
88
+
89
+ if __name__ == "__main__":
90
+ # Quick demo: pretend we have 64 GB
91
+ print("Demo: 64 GB unified memory, ~46 GB available for AI\n")
92
+ for m in recommend(available_gb=46.0):
93
+ print(" " + m.line())