llm-sizer 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +7 -6
  2. package/llm-sizer.js +111 -17
  3. package/package.json +4 -2
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # llm-sizer
2
2
 
3
- Benchmark your Apple Silicon Mac's memory bandwidth, disk speed, and MLX LLM inference throughput to find the optimal model size for your machine.
3
+ Benchmark your system's memory bandwidth, disk speed, and LLM inference throughput to find the optimal model size for your machine. Supports macOS (Apple Silicon via MLX) and Linux (via PyTorch/Transformers).
4
4
 
5
5
  ## Quick Start
6
6
 
@@ -18,18 +18,19 @@ bun llm-sizer.js
18
18
 
19
19
  ## Features
20
20
 
21
- - **Hardware Benchmarks**: Measures single-threaded and multi-threaded memory bandwidth (GB/s) along with disk write throughput.
22
- - **Model Discovery**: Scans local caches (`~/.cache/huggingface/hub`, `/tmp/mlx-models`, etc.) and fetches popular community models from `mlx-community` within your machine's unified memory budget.
21
+ - **Hardware Benchmarks**: Measures single-threaded and multi-threaded memory bandwidth (GB/s) along with disk write throughput (and detects GPU/VRAM on Linux).
22
+ - **Model Discovery**: Scans local caches (`~/.cache/huggingface/hub`, `/tmp/llm-models`, `/tmp/mlx-models`, etc.) and fetches popular community models from Hugging Face within your machine's memory budget.
23
23
  - **On-Demand Downloads**: Automatically downloads selected models from Hugging Face Hub if they aren't already cached locally.
24
- - **Inference Benchmarking**: Measures Time To First Token (TTFT), generation speed (Tokens/sec), and total tokens using `mlx-lm`.
24
+ - **Inference Benchmarking**: Measures Time To First Token (TTFT), generation speed (Tokens/sec), and total tokens using `mlx-lm` on macOS or `transformers` on Linux.
25
25
  - **Interactive Speed Demo**: Offers a live 4-second token streaming demonstration in your terminal rendered at the exact speed measured.
26
26
  - **Iterative Sizing**: Once a benchmark completes, optionally steps down by half to compare performance across model sizes.
27
27
 
28
28
  ## Requirements
29
29
 
30
- - **macOS** with Apple Silicon (M1/M2/M3/M4/M5)
30
+ - **macOS** with Apple Silicon (uses `mlx-lm`) or **Linux** (uses `transformers` and `torch`)
31
31
  - **Node.js** (>= 18) or **Bun**
32
- - **Python 3** with `mlx-lm` (prompted automatically if missing)
32
+ - **Python 3** (required packages are automatically prompted for installation if missing)
33
+ - *Note: Windows is not supported.*
33
34
 
34
35
  ## License
35
36
 
package/llm-sizer.js CHANGED
@@ -6,6 +6,11 @@ import { cpus, homedir, totalmem } from "node:os";
6
6
  import { createInterface } from "node:readline";
7
7
  import { Worker, isMainThread, parentPort, workerData } from "node:worker_threads";
8
8
 
9
+ if (process.platform === "win32") {
10
+ console.log("Not supported in Windows");
11
+ process.exit(1);
12
+ }
13
+
9
14
  if (!isMainThread) {
10
15
  const source = new Uint8Array(workerData.bytes), target = new Uint8Array(source.length);
11
16
  parentPort.on("message", ({ copies }) => {
@@ -15,6 +20,8 @@ if (!isMainThread) {
15
20
  });
16
21
  parentPort.postMessage("ready");
17
22
  } else {
23
+ const isMac = process.platform === "darwin";
24
+ const isLinux = process.platform === "linux";
18
25
  const rl = createInterface({ input: process.stdin, output: process.stdout });
19
26
  const ask = q => new Promise(res => {
20
27
  if (rl.closed) return res("");
@@ -41,12 +48,22 @@ let python;
41
48
  for (const candidate of pythonCandidates) {
42
49
  if (existsSync(candidate)) { python = candidate; break; }
43
50
  }
44
- let mlxInstalled = false;
51
+
52
+ const backendName = isMac ? "mlx-lm" : "transformers & torch";
53
+ const testCommand = isMac ? "import mlx_lm" : "import transformers, torch";
54
+ const installPrompt = isMac
55
+ ? "mlx-lm is not installed. Install mlx-lm? [y/N] "
56
+ : "transformers and torch are not installed. Install them? [y/N] ";
57
+ const installPackages = isMac
58
+ ? ["mlx-lm"]
59
+ : ["transformers", "torch", "accelerate"];
60
+
61
+ let backendInstalled = false;
45
62
  if (python) {
46
- mlxInstalled = (await spawnProc(python, ["-c", "import mlx_lm"], { stdio: "ignore" }).exited) === 0;
47
- if (!mlxInstalled && /^y(es)?$/i.test(await ask("mlx-lm is not installed. Install mlx-lm? [y/N] ") ?? "")) {
48
- const install = spawnProc(python, ["-m", "pip", "install", "--user", "--break-system-packages", "mlx-lm"]);
49
- if (await install.exited === 0) mlxInstalled = (await spawnProc(python, ["-c", "import mlx_lm"], { stdio: "ignore" }).exited) === 0;
63
+ backendInstalled = (await spawnProc(python, ["-c", testCommand], { stdio: "ignore" }).exited) === 0;
64
+ if (!backendInstalled && /^y(es)?$/i.test(await ask(installPrompt) ?? "")) {
65
+ const install = spawnProc(python, ["-m", "pip", "install", "--user", "--break-system-packages", ...installPackages]);
66
+ if (await install.exited === 0) backendInstalled = (await spawnProc(python, ["-c", testCommand], { stdio: "ignore" }).exited) === 0;
50
67
  }
51
68
  }
52
69
 
@@ -96,20 +113,34 @@ try {
96
113
  await Promise.all(workers.map(w => w.terminate()));
97
114
  }
98
115
 
116
+ let gpuInfo = "";
117
+ let gpuMem = 0;
118
+ if (!isMac && python) {
119
+ try {
120
+ const out = execSync(`${python} -c "import torch; print(f'{torch.cuda.get_device_name(0)}|{torch.cuda.get_device_properties(0).total_memory}') if torch.cuda.is_available() else print('')" 2>/dev/null`, { encoding: "utf8" }).trim();
121
+ if (out) {
122
+ const [name, mem] = out.split("|");
123
+ gpuMem = Number(mem) || 0;
124
+ gpuInfo = ` | GPU: ${name} (${(gpuMem / 1024 ** 3).toFixed(2)} GB VRAM)`;
125
+ }
126
+ } catch {}
127
+ }
128
+
99
129
  console.log("\n========== SYSTEM PERFORMANCE ==========");
100
- console.log(`CPU: ${cpus()[0]?.model ?? "Unknown"} | RAM: ${(totalmem() / 1024 ** 3).toFixed(2)} GB | Mem: ${bandwidth.toFixed(2)} GB/s | Mem MT: ${multiBandwidth.toFixed(2)} GB/s | Disk: ${storage.toFixed(2)} GB/s`);
130
+ console.log(`CPU: ${cpus()[0]?.model ?? "Unknown"} | RAM: ${(totalmem() / 1024 ** 3).toFixed(2)} GB${gpuInfo} | Mem: ${bandwidth.toFixed(2)} GB/s | Mem MT: ${multiBandwidth.toFixed(2)} GB/s | Disk: ${storage.toFixed(2)} GB/s`);
101
131
  console.log("========================================");
102
132
  console.log(`${python ? "✅" : "❌"} Python installed`);
103
- console.log(`${mlxInstalled ? "✅" : "❌"} mlx-lm installed`);
133
+ console.log(`${backendInstalled ? "✅" : "❌"} ${backendName} installed`);
104
134
 
105
- if (!python || !mlxInstalled) {
106
- console.error("\nPython 3 and mlx-lm are required to run model benchmarks.");
135
+ if (!python || !backendInstalled) {
136
+ console.error(`\nPython 3 and ${backendName} are required to run model benchmarks.`);
107
137
  rl.close();
108
138
  process.exit(1);
109
139
  }
110
140
 
111
141
  const modelSize = name => {
112
142
  if (/-vl\b|-vision\b|whisper/i.test(name)) return 0;
143
+ if (!isMac && /-(?:gguf|awq|gptq|exl2)\b/i.test(name)) return 0;
113
144
  const p = name.match(/(\d+(?:\.\d+)?)\s*([bm])(?:-|$)/i);
114
145
  const b = name.match(/(?:-|\b)(\d+)-?bit/i);
115
146
  if (!p) return 0;
@@ -127,7 +158,9 @@ const confirmDiskSpace = async (directory, requiredBytes) => {
127
158
  };
128
159
 
129
160
  const localModels = [];
161
+ const modelDownloadDir = isMac ? "/tmp/mlx-models" : "/tmp/llm-models";
130
162
  const roots = [
163
+ modelDownloadDir,
131
164
  "/tmp/mlx-models",
132
165
  `${homedir()}/.cache/huggingface/hub`,
133
166
  `${homedir()}/models`,
@@ -152,10 +185,13 @@ const scan = async (directory, depth = 0) => {
152
185
  };
153
186
  for (const root of roots) await scan(root);
154
187
 
155
- const budget = totalmem() * 0.65;
188
+ const budget = (gpuMem > 0 ? gpuMem : totalmem()) * 0.65;
156
189
  const candidates = localModels.filter(m => m.bytes <= budget);
190
+ const hfApiUrl = isMac
191
+ ? "https://huggingface.co/api/models?author=mlx-community&search=Instruct&sort=downloads&direction=-1&limit=100"
192
+ : "https://huggingface.co/api/models?pipeline_tag=text-generation&search=Instruct&sort=downloads&direction=-1&limit=100";
157
193
  try {
158
- const response = await fetch("https://huggingface.co/api/models?author=mlx-community&search=Instruct&sort=downloads&direction=-1&limit=100", { signal: AbortSignal.timeout(8000) });
194
+ const response = await fetch(hfApiUrl, { signal: AbortSignal.timeout(8000) });
159
195
  if (response.ok) {
160
196
  for (const model of await response.json()) {
161
197
  const bytes = modelSize(model.id);
@@ -177,7 +213,7 @@ for (;;) {
177
213
 
178
214
  let selectedModel = model.local ? model.path : null;
179
215
  if (!model.local) {
180
- const modelPath = `/tmp/mlx-models/${model.name.replaceAll("/", "--")}`;
216
+ const modelPath = `${modelDownloadDir}/${model.name.replaceAll("/", "--")}`;
181
217
  if (await confirmDiskSpace("/tmp", model.bytes * 1.1)) {
182
218
  const folderSize = async dir => {
183
219
  let bytes = 0;
@@ -214,7 +250,7 @@ for (;;) {
214
250
  }
215
251
  if (!selectedModel) break;
216
252
 
217
- const benchmark = `
253
+ const benchmark = isMac ? `
218
254
  import json, os, time
219
255
  from mlx_lm import load, stream_generate
220
256
 
@@ -235,8 +271,61 @@ for prompt in prompts:
235
271
  if last_resp and last_resp.generation_tokens:
236
272
  ttft = (first - start) if first else 0
237
273
  print(json.dumps({"ttft": ttft, "tps": last_resp.generation_tps, "tokens": last_resp.generation_tokens}), flush=True)
274
+ ` : `
275
+ import json, os, time, warnings
276
+ warnings.filterwarnings("ignore")
277
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
278
+ import torch
279
+ from transformers import AutoModelForCausalLM, AutoTokenizer
280
+ from transformers.generation.streamers import BaseStreamer
281
+
282
+ model_path = os.environ["MODEL_PATH"]
283
+ device = "cuda" if torch.cuda.is_available() else "cpu"
284
+ dtype = torch.float16 if torch.cuda.is_available() else torch.float32
285
+
286
+ tokenizer = AutoTokenizer.from_pretrained(model_path)
287
+ if tokenizer.pad_token_id is None:
288
+ tokenizer.pad_token_id = tokenizer.eos_token_id
289
+ model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=dtype, low_cpu_mem_usage=True).to(device)
290
+
291
+ class MetricStreamer(BaseStreamer):
292
+ def __init__(self, start):
293
+ self.start = start
294
+ self.first = None
295
+ self.last = start
296
+ self.count = 0
297
+ self.is_prompt = True
298
+ def put(self, value):
299
+ if self.is_prompt:
300
+ self.is_prompt = False
301
+ return
302
+ now = time.perf_counter()
303
+ if self.first is None: self.first = now
304
+ self.count += 1
305
+ self.last = now
306
+ def end(self): pass
307
+
308
+ prompts = [
309
+ "Explain what RAM does in about 50 words.",
310
+ "Explain why the sky appears blue in about 50 words.",
311
+ "Give a simple three-step recipe for making tea.",
312
+ "Explain the difference between a CPU and a GPU in about 50 words.",
313
+ "Describe the water cycle in about 50 words."
314
+ ]
315
+ for prompt in prompts:
316
+ inputs = tokenizer(prompt, return_tensors="pt").to(device)
317
+ start = time.perf_counter()
318
+ streamer = MetricStreamer(start)
319
+ model.generate(**inputs, streamer=streamer, max_new_tokens=128)
320
+ if streamer.count:
321
+ ttft = (streamer.first - start) if streamer.first else 0
322
+ gen_time = (streamer.last - streamer.first) if (streamer.first and streamer.last > streamer.first) else (streamer.last - start)
323
+ tps = streamer.count / gen_time if gen_time > 0 else 0
324
+ print(json.dumps({"ttft": ttft, "tps": tps, "tokens": streamer.count}), flush=True)
238
325
  `;
239
- console.log(`\nRunning MLX benchmark for ${model.name}...`);
326
+
327
+ const engineName = isMac ? "MLX" : "Transformers";
328
+ console.log(`\nRunning ${engineName} benchmark for ${model.name}...`);
240
329
  const test = spawnProc(python, ["-c", benchmark], {
241
330
  env: { ...process.env, MODEL_PATH: selectedModel },
242
331
  capture: true
@@ -244,15 +333,20 @@ for prompt in prompts:
244
333
  const benchmarkSuccess = (await test.exited) === 0;
245
334
  const output = test.getOutput();
246
335
  if (benchmarkSuccess) {
247
- const results = output.trim().split("\n").filter(Boolean).map(JSON.parse);
336
+ const results = output.trim().split("\n").map(l => {
337
+ try { return JSON.parse(l.trim()); } catch { return null; }
338
+ }).filter(Boolean);
248
339
  if (results.length) {
249
340
  const meanTps = results.reduce((s, r) => s + r.tps, 0) / results.length;
250
- console.log("\n========== MLX TOKEN TEST ==========");
341
+ console.log(`\n========== ${engineName.toUpperCase()} TOKEN TEST ==========`);
251
342
  results.forEach((r, i) => console.log(`${i + 1}. TTFT: ${(r.ttft * 1000).toFixed(0)} ms | Tok/s: ${r.tps.toFixed(2)} | Tokens: ${r.tokens}`));
252
343
  console.log(`Mean TTFT: ${(results.reduce((s, r) => s + r.ttft, 0) / results.length * 1000).toFixed(0)} ms | Mean Tok/s: ${meanTps.toFixed(2)}`);
253
344
  console.log("=====================================");
254
345
  if (/^y(es)?$/i.test(await ask(`Show how fast ${meanTps.toFixed(1)} tok/s looks like? [y/N] `) ?? "")) {
255
- const words = "Artificial intelligence and large language models process text by predicting the next token in a sequence. With Apple Silicon unified memory architecture, weights are streamed with high bandwidth directly to the GPU cores. ".repeat(5).match(/\s*\S+/g);
346
+ const demoText = isMac
347
+ ? "Artificial intelligence and large language models process text by predicting the next token in a sequence. With Apple Silicon unified memory architecture, weights are streamed with high bandwidth directly to the GPU cores. "
348
+ : "Artificial intelligence and large language models process text by predicting the next token in a sequence. Weights and activations are streamed with high bandwidth directly to the compute cores for fast token generation. ";
349
+ const words = demoText.repeat(5).match(/\s*\S+/g);
256
350
  const demoStart = performance.now();
257
351
  for (let i = 0; performance.now() - demoStart < 4000; i++) {
258
352
  process.stdout.write(words[i % words.length]);
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "llm-sizer",
3
- "version": "0.1.0",
4
- "description": "Benchmark Apple Silicon memory bandwidth, disk speed, and MLX model inference performance to find the optimal LLM size",
3
+ "version": "0.1.1",
4
+ "description": "Benchmark memory bandwidth, disk speed, and LLM inference performance (MLX on macOS, Transformers on Linux) to find the optimal model size",
5
5
  "type": "module",
6
6
  "main": "llm-sizer.js",
7
7
  "bin": {
@@ -17,10 +17,12 @@
17
17
  },
18
18
  "keywords": [
19
19
  "mlx",
20
+ "transformers",
20
21
  "benchmark",
21
22
  "apple-silicon",
22
23
  "llm",
23
24
  "mac",
25
+ "linux",
24
26
  "inference"
25
27
  ],
26
28
  "license": "MIT"