llm-sizer 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -0
- package/llm-sizer.js +274 -0
- package/package.json +27 -0
package/README.md
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# llm-sizer
|
|
2
|
+
|
|
3
|
+
Benchmark your Apple Silicon Mac's memory bandwidth, disk speed, and MLX LLM inference throughput to find the optimal model size for your machine.
|
|
4
|
+
|
|
5
|
+
## Quick Start
|
|
6
|
+
|
|
7
|
+
Run instantly without installing:
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
npx llm-sizer
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Or with [Bun](https://bun.sh):
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
bun llm-sizer.js
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Features
|
|
20
|
+
|
|
21
|
+
- **Hardware Benchmarks**: Measures single-threaded and multi-threaded memory bandwidth (GB/s) along with disk write throughput.
|
|
22
|
+
- **Model Discovery**: Scans local caches (`~/.cache/huggingface/hub`, `/tmp/mlx-models`, etc.) and fetches popular community models from `mlx-community` within your machine's unified memory budget.
|
|
23
|
+
- **On-Demand Downloads**: Automatically downloads selected models from Hugging Face Hub if they aren't already cached locally.
|
|
24
|
+
- **Inference Benchmarking**: Measures Time To First Token (TTFT), generation speed (Tokens/sec), and total tokens using `mlx-lm`.
|
|
25
|
+
- **Interactive Speed Demo**: Offers a live 4-second token streaming demonstration in your terminal rendered at the exact speed measured.
|
|
26
|
+
- **Iterative Sizing**: Once a benchmark completes, optionally steps down by half to compare performance across model sizes.
|
|
27
|
+
|
|
28
|
+
## Requirements
|
|
29
|
+
|
|
30
|
+
- **macOS** with Apple Silicon (M1/M2/M3/M4/M5)
|
|
31
|
+
- **Node.js** (>= 18) or **Bun**
|
|
32
|
+
- **Python 3** with `mlx-lm` (prompted automatically if missing)
|
|
33
|
+
|
|
34
|
+
## License
|
|
35
|
+
|
|
36
|
+
MIT
|
package/llm-sizer.js
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { execSync, spawn } from "node:child_process";
|
|
3
|
+
import { existsSync } from "node:fs";
|
|
4
|
+
import { open, readdir, rm, stat, statfs, unlink } from "node:fs/promises";
|
|
5
|
+
import { cpus, homedir, totalmem } from "node:os";
|
|
6
|
+
import { createInterface } from "node:readline";
|
|
7
|
+
import { Worker, isMainThread, parentPort, workerData } from "node:worker_threads";
|
|
8
|
+
|
|
9
|
+
if (!isMainThread) {
|
|
10
|
+
const source = new Uint8Array(workerData.bytes), target = new Uint8Array(source.length);
|
|
11
|
+
parentPort.on("message", ({ copies }) => {
|
|
12
|
+
const start = performance.now();
|
|
13
|
+
for (let i = 0; i < copies; i++) target.set(source);
|
|
14
|
+
parentPort.postMessage(target.length * copies / ((performance.now() - start) / 1000) / 1e9);
|
|
15
|
+
});
|
|
16
|
+
parentPort.postMessage("ready");
|
|
17
|
+
} else {
|
|
18
|
+
const rl = createInterface({ input: process.stdin, output: process.stdout });
|
|
19
|
+
const ask = q => new Promise(res => {
|
|
20
|
+
if (rl.closed) return res("");
|
|
21
|
+
rl.question(q, res);
|
|
22
|
+
rl.once("close", () => res(""));
|
|
23
|
+
});
|
|
24
|
+
const sleep = ms => new Promise(r => setTimeout(r, ms));
|
|
25
|
+
const which = cmd => {
|
|
26
|
+
try { return execSync(`which ${cmd} 2>/dev/null`, { encoding: "utf8" }).trim(); } catch { return null; }
|
|
27
|
+
};
|
|
28
|
+
const spawnProc = (cmd, args, { env = process.env, stdio = "inherit", capture = false } = {}) => {
|
|
29
|
+
const cp = spawn(cmd, args, { env, stdio: capture ? ["inherit", "pipe", "inherit"] : stdio });
|
|
30
|
+
let stdout = "";
|
|
31
|
+
if (capture && cp.stdout) cp.stdout.on("data", d => { stdout += d; });
|
|
32
|
+
return { exited: new Promise(res => cp.on("close", res)), kill: sig => cp.kill(sig), getOutput: () => stdout };
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
const pythonCandidates = [
|
|
36
|
+
process.env.VIRTUAL_ENV && `${process.env.VIRTUAL_ENV}/bin/python`,
|
|
37
|
+
which("python3"),
|
|
38
|
+
which("python")
|
|
39
|
+
].filter(Boolean);
|
|
40
|
+
let python;
|
|
41
|
+
for (const candidate of pythonCandidates) {
|
|
42
|
+
if (existsSync(candidate)) { python = candidate; break; }
|
|
43
|
+
}
|
|
44
|
+
let mlxInstalled = false;
|
|
45
|
+
if (python) {
|
|
46
|
+
mlxInstalled = (await spawnProc(python, ["-c", "import mlx_lm"], { stdio: "ignore" }).exited) === 0;
|
|
47
|
+
if (!mlxInstalled && /^y(es)?$/i.test(await ask("mlx-lm is not installed. Install mlx-lm? [y/N] ") ?? "")) {
|
|
48
|
+
const install = spawnProc(python, ["-m", "pip", "install", "--user", "--break-system-packages", "mlx-lm"]);
|
|
49
|
+
if (await install.exited === 0) mlxInstalled = (await spawnProc(python, ["-c", "import mlx_lm"], { stdio: "ignore" }).exited) === 0;
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const source = new Uint8Array(256 * 1024 ** 2), target = new Uint8Array(source.length);
|
|
54
|
+
target.set(source);
|
|
55
|
+
let bandwidth = 0;
|
|
56
|
+
for (let run = 0; run < 5; run++) {
|
|
57
|
+
const start = performance.now();
|
|
58
|
+
for (let i = 0; i < 4; i++) target.set(source);
|
|
59
|
+
bandwidth += source.byteLength * 4 / ((performance.now() - start) / 1000) / 1e9;
|
|
60
|
+
}
|
|
61
|
+
bandwidth /= 5;
|
|
62
|
+
|
|
63
|
+
const storageFile = `/tmp/bun-storage-${process.pid}.bin`, storageData = new Uint8Array(64 * 1024 ** 2);
|
|
64
|
+
storageData.fill(0xa5);
|
|
65
|
+
let storage = 0;
|
|
66
|
+
try {
|
|
67
|
+
const file = await open(storageFile, "w");
|
|
68
|
+
try {
|
|
69
|
+
for (let run = 0; run < 5; run++) {
|
|
70
|
+
const start = performance.now();
|
|
71
|
+
await file.write(storageData, 0, storageData.length, 0);
|
|
72
|
+
await file.sync();
|
|
73
|
+
storage += storageData.length / ((performance.now() - start) / 1000) / 1e9;
|
|
74
|
+
}
|
|
75
|
+
} finally { await file.close(); }
|
|
76
|
+
} finally { await unlink(storageFile).catch(() => {}); }
|
|
77
|
+
storage /= 5;
|
|
78
|
+
|
|
79
|
+
let multiBandwidth = 0;
|
|
80
|
+
const workers = Array.from({ length: cpus().length }, () => new Worker(new URL(import.meta.url), {
|
|
81
|
+
workerData: { bytes: 32 * 1024 ** 2 }
|
|
82
|
+
}));
|
|
83
|
+
try {
|
|
84
|
+
await Promise.all(workers.map(w => new Promise((res, rej) => { w.once("message", res); w.once("error", rej); })));
|
|
85
|
+
for (let run = 0; run < 5; run++) {
|
|
86
|
+
const start = performance.now();
|
|
87
|
+
await Promise.all(workers.map(w => new Promise((res, rej) => {
|
|
88
|
+
w.once("message", res);
|
|
89
|
+
w.once("error", rej);
|
|
90
|
+
w.postMessage({ copies: 4 });
|
|
91
|
+
})));
|
|
92
|
+
multiBandwidth += cpus().length * 32 * 1024 ** 2 * 4 / ((performance.now() - start) / 1000) / 1e9;
|
|
93
|
+
}
|
|
94
|
+
multiBandwidth /= 5;
|
|
95
|
+
} finally {
|
|
96
|
+
await Promise.all(workers.map(w => w.terminate()));
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
console.log("\n========== SYSTEM PERFORMANCE ==========");
|
|
100
|
+
console.log(`CPU: ${cpus()[0]?.model ?? "Unknown"} | RAM: ${(totalmem() / 1024 ** 3).toFixed(2)} GB | Mem: ${bandwidth.toFixed(2)} GB/s | Mem MT: ${multiBandwidth.toFixed(2)} GB/s | Disk: ${storage.toFixed(2)} GB/s`);
|
|
101
|
+
console.log("========================================");
|
|
102
|
+
console.log(`${python ? "✅" : "❌"} Python installed`);
|
|
103
|
+
console.log(`${mlxInstalled ? "✅" : "❌"} mlx-lm installed`);
|
|
104
|
+
|
|
105
|
+
if (!python || !mlxInstalled) {
|
|
106
|
+
console.error("\nPython 3 and mlx-lm are required to run model benchmarks.");
|
|
107
|
+
rl.close();
|
|
108
|
+
process.exit(1);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const modelSize = name => {
|
|
112
|
+
if (/-vl\b|-vision\b|whisper/i.test(name)) return 0;
|
|
113
|
+
const p = name.match(/(\d+(?:\.\d+)?)\s*([bm])(?:-|$)/i);
|
|
114
|
+
const b = name.match(/(?:-|\b)(\d+)-?bit/i);
|
|
115
|
+
if (!p) return 0;
|
|
116
|
+
const params = Number(p[1]) * (p[2].toUpperCase() === "B" ? 1e9 : 1e6);
|
|
117
|
+
const bits = b ? Number(b[1]) : 16;
|
|
118
|
+
return params * (bits / 8) * 1.1;
|
|
119
|
+
};
|
|
120
|
+
const sizeText = bytes => bytes >= 1e9 ? `${(bytes / 1e9).toFixed(1)} GB` : `${(bytes / 1e6).toFixed(0)} MB`;
|
|
121
|
+
const confirmDiskSpace = async (directory, requiredBytes) => {
|
|
122
|
+
const { bavail, bsize } = await statfs(directory);
|
|
123
|
+
const availableBytes = bavail * bsize;
|
|
124
|
+
if (availableBytes >= requiredBytes) return true;
|
|
125
|
+
console.error(`Not enough disk space for ${sizeText(requiredBytes)} download: ${sizeText(availableBytes)} available in ${directory}.`);
|
|
126
|
+
return /^y(es)?$/i.test(await ask("Free some disk space, or continue anyway? [y/N] ") ?? "");
|
|
127
|
+
};
|
|
128
|
+
|
|
129
|
+
const localModels = [];
|
|
130
|
+
const roots = [
|
|
131
|
+
"/tmp/mlx-models",
|
|
132
|
+
`${homedir()}/.cache/huggingface/hub`,
|
|
133
|
+
`${homedir()}/models`,
|
|
134
|
+
`${homedir()}/Models`
|
|
135
|
+
];
|
|
136
|
+
const scan = async (directory, depth = 0) => {
|
|
137
|
+
if (depth > 4 || /whisper/i.test(directory)) return;
|
|
138
|
+
let entries;
|
|
139
|
+
try { entries = await readdir(directory, { withFileTypes: true }); } catch { return; }
|
|
140
|
+
const weights = entries.filter(e => /\.(safetensors|bin|gguf)$/i.test(e.name));
|
|
141
|
+
if (entries.some(e => e.name === "config.json") && weights.length) {
|
|
142
|
+
let bytes = 0;
|
|
143
|
+
for (const w of weights) {
|
|
144
|
+
try { bytes += (await stat(`${directory}/${w.name}`)).size; } catch {}
|
|
145
|
+
}
|
|
146
|
+
const hubMatch = directory.match(/models--([^/]+)/);
|
|
147
|
+
const name = hubMatch ? hubMatch[1].replaceAll("--", "/") : directory.split("/").pop().replaceAll("--", "/");
|
|
148
|
+
if (!localModels.some(m => m.name === name)) localModels.push({ name, path: directory, bytes, local: true });
|
|
149
|
+
return;
|
|
150
|
+
}
|
|
151
|
+
for (const e of entries.filter(e => e.isDirectory())) await scan(`${directory}/${e.name}`, depth + 1);
|
|
152
|
+
};
|
|
153
|
+
for (const root of roots) await scan(root);
|
|
154
|
+
|
|
155
|
+
const budget = totalmem() * 0.65;
|
|
156
|
+
const candidates = localModels.filter(m => m.bytes <= budget);
|
|
157
|
+
try {
|
|
158
|
+
const response = await fetch("https://huggingface.co/api/models?author=mlx-community&search=Instruct&sort=downloads&direction=-1&limit=100", { signal: AbortSignal.timeout(8000) });
|
|
159
|
+
if (response.ok) {
|
|
160
|
+
for (const model of await response.json()) {
|
|
161
|
+
const bytes = modelSize(model.id);
|
|
162
|
+
if (bytes && bytes <= budget && !candidates.some(c => c.name === model.id)) candidates.push({ name: model.id, bytes });
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
} catch {}
|
|
166
|
+
|
|
167
|
+
let pool;
|
|
168
|
+
for (;;) {
|
|
169
|
+
const choices = (pool ?? candidates).sort((a, b) => b.bytes - a.bytes).slice(0, 4);
|
|
170
|
+
if (!choices.length) break;
|
|
171
|
+
console.log("\n========== MODEL SELECTION ==========");
|
|
172
|
+
choices.forEach((m, i) => console.log(`${i + 1}. ${m.name} (${sizeText(m.bytes)}${m.local ? ", local" : ""})`));
|
|
173
|
+
console.log("=====================================");
|
|
174
|
+
const selected = Number(await ask("Choose a model to use, or press Enter to skip: ")) - 1;
|
|
175
|
+
const model = choices[selected];
|
|
176
|
+
if (!model) break;
|
|
177
|
+
|
|
178
|
+
let selectedModel = model.local ? model.path : null;
|
|
179
|
+
if (!model.local) {
|
|
180
|
+
const modelPath = `/tmp/mlx-models/${model.name.replaceAll("/", "--")}`;
|
|
181
|
+
if (await confirmDiskSpace("/tmp", model.bytes * 1.1)) {
|
|
182
|
+
const folderSize = async dir => {
|
|
183
|
+
let bytes = 0;
|
|
184
|
+
const walk = async p => {
|
|
185
|
+
let list;
|
|
186
|
+
try { list = await readdir(p, { withFileTypes: true }); } catch { return; }
|
|
187
|
+
for (const item of list) {
|
|
188
|
+
const sub = `${p}/${item.name}`;
|
|
189
|
+
if (item.isDirectory()) await walk(sub);
|
|
190
|
+
else try { bytes += (await stat(sub)).size; } catch {}
|
|
191
|
+
}
|
|
192
|
+
};
|
|
193
|
+
await walk(dir);
|
|
194
|
+
return bytes;
|
|
195
|
+
};
|
|
196
|
+
const token = process.env.HF_TOKEN ?? process.env.HF_API_KEY ?? process.env.HUGGINGFACE_API_KEY ?? process.env.HUGGING_FACE_HUB_TOKEN ?? ((await ask("No HF token in env. Enter a Hugging Face token, or press Enter to continue anonymously: ") ?? "") || undefined);
|
|
197
|
+
const download = spawnProc(python, ["-c", "from huggingface_hub import snapshot_download; snapshot_download(repo_id=__import__('os').environ['MODEL_ID'], local_dir=__import__('os').environ['MODEL_DIR'])"], {
|
|
198
|
+
env: { ...process.env, MODEL_ID: model.name, MODEL_DIR: modelPath, ...(token ? { HF_TOKEN: token, HUGGING_FACE_HUB_TOKEN: token } : {}) },
|
|
199
|
+
stdio: ["ignore", "ignore", "inherit"]
|
|
200
|
+
});
|
|
201
|
+
process.stdout.write(`Downloading ${model.name}...`);
|
|
202
|
+
const progress = setInterval(async () => process.stdout.write(`\rDownloading ${model.name}: ${sizeText(await folderSize(modelPath))}`), 1000);
|
|
203
|
+
const onSigint = async () => { clearInterval(progress); download.kill("SIGINT"); await rm(modelPath, { recursive: true, force: true }); process.exit(130); };
|
|
204
|
+
process.on("SIGINT", onSigint);
|
|
205
|
+
const exitCode = await download.exited;
|
|
206
|
+
process.off("SIGINT", onSigint);
|
|
207
|
+
clearInterval(progress);
|
|
208
|
+
process.stdout.write(`\r${exitCode === 0 ? "Downloaded" : "Download failed"}: ${model.name}\n`);
|
|
209
|
+
if (exitCode === 0) {
|
|
210
|
+
selectedModel = modelPath;
|
|
211
|
+
console.log(`Model path: ${modelPath}`);
|
|
212
|
+
} else await rm(modelPath, { recursive: true, force: true });
|
|
213
|
+
} else console.log("Download skipped.");
|
|
214
|
+
}
|
|
215
|
+
if (!selectedModel) break;
|
|
216
|
+
|
|
217
|
+
const benchmark = `
|
|
218
|
+
import json, os, time
|
|
219
|
+
from mlx_lm import load, stream_generate
|
|
220
|
+
|
|
221
|
+
model, tokenizer = load(os.environ["MODEL_PATH"])
|
|
222
|
+
prompts = [
|
|
223
|
+
"Explain what RAM does in about 50 words.",
|
|
224
|
+
"Explain why the sky appears blue in about 50 words.",
|
|
225
|
+
"Give a simple three-step recipe for making tea.",
|
|
226
|
+
"Explain the difference between a CPU and a GPU in about 50 words.",
|
|
227
|
+
"Describe the water cycle in about 50 words."
|
|
228
|
+
]
|
|
229
|
+
for prompt in prompts:
|
|
230
|
+
start = time.perf_counter()
|
|
231
|
+
first, last_resp = None, None
|
|
232
|
+
for response in stream_generate(model, tokenizer, prompt, max_tokens=128):
|
|
233
|
+
if first is None: first = time.perf_counter()
|
|
234
|
+
last_resp = response
|
|
235
|
+
if last_resp and last_resp.generation_tokens:
|
|
236
|
+
ttft = (first - start) if first else 0
|
|
237
|
+
print(json.dumps({"ttft": ttft, "tps": last_resp.generation_tps, "tokens": last_resp.generation_tokens}), flush=True)
|
|
238
|
+
`;
|
|
239
|
+
console.log(`\nRunning MLX benchmark for ${model.name}...`);
|
|
240
|
+
const test = spawnProc(python, ["-c", benchmark], {
|
|
241
|
+
env: { ...process.env, MODEL_PATH: selectedModel },
|
|
242
|
+
capture: true
|
|
243
|
+
});
|
|
244
|
+
const benchmarkSuccess = (await test.exited) === 0;
|
|
245
|
+
const output = test.getOutput();
|
|
246
|
+
if (benchmarkSuccess) {
|
|
247
|
+
const results = output.trim().split("\n").filter(Boolean).map(JSON.parse);
|
|
248
|
+
if (results.length) {
|
|
249
|
+
const meanTps = results.reduce((s, r) => s + r.tps, 0) / results.length;
|
|
250
|
+
console.log("\n========== MLX TOKEN TEST ==========");
|
|
251
|
+
results.forEach((r, i) => console.log(`${i + 1}. TTFT: ${(r.ttft * 1000).toFixed(0)} ms | Tok/s: ${r.tps.toFixed(2)} | Tokens: ${r.tokens}`));
|
|
252
|
+
console.log(`Mean TTFT: ${(results.reduce((s, r) => s + r.ttft, 0) / results.length * 1000).toFixed(0)} ms | Mean Tok/s: ${meanTps.toFixed(2)}`);
|
|
253
|
+
console.log("=====================================");
|
|
254
|
+
if (/^y(es)?$/i.test(await ask(`Show how fast ${meanTps.toFixed(1)} tok/s looks like? [y/N] `) ?? "")) {
|
|
255
|
+
const words = "Artificial intelligence and large language models process text by predicting the next token in a sequence. With Apple Silicon unified memory architecture, weights are streamed with high bandwidth directly to the GPU cores. ".repeat(5).match(/\s*\S+/g);
|
|
256
|
+
const demoStart = performance.now();
|
|
257
|
+
for (let i = 0; performance.now() - demoStart < 4000; i++) {
|
|
258
|
+
process.stdout.write(words[i % words.length]);
|
|
259
|
+
await sleep(1000 / meanTps);
|
|
260
|
+
}
|
|
261
|
+
console.log("\n");
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
} else {
|
|
265
|
+
console.error(`\nBenchmark of ${model.name} failed.`);
|
|
266
|
+
break;
|
|
267
|
+
}
|
|
268
|
+
const half = candidates.filter(c => c.bytes && c.bytes <= model.bytes / 2);
|
|
269
|
+
const nextPool = half.length ? half : candidates.filter(c => c.bytes && c.bytes > 0 && c.bytes < model.bytes);
|
|
270
|
+
if (!nextPool.length || !/^y(es)?$/i.test(await ask(`\nBenchmark of ${model.name} done. Try a model about half the size? [y/N] `) ?? "")) break;
|
|
271
|
+
pool = nextPool;
|
|
272
|
+
}
|
|
273
|
+
rl.close();
|
|
274
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "llm-sizer",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Benchmark Apple Silicon memory bandwidth, disk speed, and MLX model inference performance to find the optimal LLM size",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "llm-sizer.js",
|
|
7
|
+
"bin": {
|
|
8
|
+
"llm-sizer": "llm-sizer.js",
|
|
9
|
+
"size-me": "llm-sizer.js"
|
|
10
|
+
},
|
|
11
|
+
"files": [
|
|
12
|
+
"llm-sizer.js",
|
|
13
|
+
"README.md"
|
|
14
|
+
],
|
|
15
|
+
"engines": {
|
|
16
|
+
"node": ">=18"
|
|
17
|
+
},
|
|
18
|
+
"keywords": [
|
|
19
|
+
"mlx",
|
|
20
|
+
"benchmark",
|
|
21
|
+
"apple-silicon",
|
|
22
|
+
"llm",
|
|
23
|
+
"mac",
|
|
24
|
+
"inference"
|
|
25
|
+
],
|
|
26
|
+
"license": "MIT"
|
|
27
|
+
}
|