llm-sizer 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -6
- package/llm-sizer.js +111 -17
- package/package.json +4 -2
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# llm-sizer
|
|
2
2
|
|
|
3
|
-
Benchmark your
|
|
3
|
+
Benchmark your system's memory bandwidth, disk speed, and LLM inference throughput to find the optimal model size for your machine. Supports macOS (Apple Silicon via MLX) and Linux (via PyTorch/Transformers).
|
|
4
4
|
|
|
5
5
|
## Quick Start
|
|
6
6
|
|
|
@@ -18,18 +18,19 @@ bun llm-sizer.js
|
|
|
18
18
|
|
|
19
19
|
## Features
|
|
20
20
|
|
|
21
|
-
- **Hardware Benchmarks**: Measures single-threaded and multi-threaded memory bandwidth (GB/s) along with disk write throughput.
|
|
22
|
-
- **Model Discovery**: Scans local caches (`~/.cache/huggingface/hub`, `/tmp/mlx-models`, etc.) and fetches popular community models from
|
|
21
|
+
- **Hardware Benchmarks**: Measures single-threaded and multi-threaded memory bandwidth (GB/s) along with disk write throughput (and detects GPU/VRAM on Linux).
|
|
22
|
+
- **Model Discovery**: Scans local caches (`~/.cache/huggingface/hub`, `/tmp/llm-models`, `/tmp/mlx-models`, etc.) and fetches popular community models from Hugging Face within your machine's memory budget.
|
|
23
23
|
- **On-Demand Downloads**: Automatically downloads selected models from Hugging Face Hub if they aren't already cached locally.
|
|
24
|
-
- **Inference Benchmarking**: Measures Time To First Token (TTFT), generation speed (Tokens/sec), and total tokens using `mlx-lm
|
|
24
|
+
- **Inference Benchmarking**: Measures Time To First Token (TTFT), generation speed (Tokens/sec), and total tokens using `mlx-lm` on macOS or `transformers` on Linux.
|
|
25
25
|
- **Interactive Speed Demo**: Offers a live 4-second token streaming demonstration in your terminal rendered at the exact speed measured.
|
|
26
26
|
- **Iterative Sizing**: Once a benchmark completes, optionally steps down by half to compare performance across model sizes.
|
|
27
27
|
|
|
28
28
|
## Requirements
|
|
29
29
|
|
|
30
|
-
- **macOS** with Apple Silicon (
|
|
30
|
+
- **macOS** with Apple Silicon (uses `mlx-lm`) or **Linux** (uses `transformers` and `torch`)
|
|
31
31
|
- **Node.js** (>= 18) or **Bun**
|
|
32
|
-
- **Python 3**
|
|
32
|
+
- **Python 3** (required packages are automatically prompted for installation if missing)
|
|
33
|
+
- *Note: Windows is not supported.*
|
|
33
34
|
|
|
34
35
|
## License
|
|
35
36
|
|
package/llm-sizer.js
CHANGED
|
@@ -6,6 +6,11 @@ import { cpus, homedir, totalmem } from "node:os";
|
|
|
6
6
|
import { createInterface } from "node:readline";
|
|
7
7
|
import { Worker, isMainThread, parentPort, workerData } from "node:worker_threads";
|
|
8
8
|
|
|
9
|
+
if (process.platform === "win32") {
|
|
10
|
+
console.log("Not supported in Windows");
|
|
11
|
+
process.exit(1);
|
|
12
|
+
}
|
|
13
|
+
|
|
9
14
|
if (!isMainThread) {
|
|
10
15
|
const source = new Uint8Array(workerData.bytes), target = new Uint8Array(source.length);
|
|
11
16
|
parentPort.on("message", ({ copies }) => {
|
|
@@ -15,6 +20,8 @@ if (!isMainThread) {
|
|
|
15
20
|
});
|
|
16
21
|
parentPort.postMessage("ready");
|
|
17
22
|
} else {
|
|
23
|
+
const isMac = process.platform === "darwin";
|
|
24
|
+
const isLinux = process.platform === "linux";
|
|
18
25
|
const rl = createInterface({ input: process.stdin, output: process.stdout });
|
|
19
26
|
const ask = q => new Promise(res => {
|
|
20
27
|
if (rl.closed) return res("");
|
|
@@ -41,12 +48,22 @@ let python;
|
|
|
41
48
|
for (const candidate of pythonCandidates) {
|
|
42
49
|
if (existsSync(candidate)) { python = candidate; break; }
|
|
43
50
|
}
|
|
44
|
-
|
|
51
|
+
|
|
52
|
+
const backendName = isMac ? "mlx-lm" : "transformers & torch";
|
|
53
|
+
const testCommand = isMac ? "import mlx_lm" : "import transformers, torch";
|
|
54
|
+
const installPrompt = isMac
|
|
55
|
+
? "mlx-lm is not installed. Install mlx-lm? [y/N] "
|
|
56
|
+
: "transformers and torch are not installed. Install them? [y/N] ";
|
|
57
|
+
const installPackages = isMac
|
|
58
|
+
? ["mlx-lm"]
|
|
59
|
+
: ["transformers", "torch", "accelerate"];
|
|
60
|
+
|
|
61
|
+
let backendInstalled = false;
|
|
45
62
|
if (python) {
|
|
46
|
-
|
|
47
|
-
if (!
|
|
48
|
-
const install = spawnProc(python, ["-m", "pip", "install", "--user", "--break-system-packages",
|
|
49
|
-
if (await install.exited === 0)
|
|
63
|
+
backendInstalled = (await spawnProc(python, ["-c", testCommand], { stdio: "ignore" }).exited) === 0;
|
|
64
|
+
if (!backendInstalled && /^y(es)?$/i.test(await ask(installPrompt) ?? "")) {
|
|
65
|
+
const install = spawnProc(python, ["-m", "pip", "install", "--user", "--break-system-packages", ...installPackages]);
|
|
66
|
+
if (await install.exited === 0) backendInstalled = (await spawnProc(python, ["-c", testCommand], { stdio: "ignore" }).exited) === 0;
|
|
50
67
|
}
|
|
51
68
|
}
|
|
52
69
|
|
|
@@ -96,20 +113,34 @@ try {
|
|
|
96
113
|
await Promise.all(workers.map(w => w.terminate()));
|
|
97
114
|
}
|
|
98
115
|
|
|
116
|
+
let gpuInfo = "";
|
|
117
|
+
let gpuMem = 0;
|
|
118
|
+
if (!isMac && python) {
|
|
119
|
+
try {
|
|
120
|
+
const out = execSync(`${python} -c "import torch; print(f'{torch.cuda.get_device_name(0)}|{torch.cuda.get_device_properties(0).total_memory}') if torch.cuda.is_available() else print('')" 2>/dev/null`, { encoding: "utf8" }).trim();
|
|
121
|
+
if (out) {
|
|
122
|
+
const [name, mem] = out.split("|");
|
|
123
|
+
gpuMem = Number(mem) || 0;
|
|
124
|
+
gpuInfo = ` | GPU: ${name} (${(gpuMem / 1024 ** 3).toFixed(2)} GB VRAM)`;
|
|
125
|
+
}
|
|
126
|
+
} catch {}
|
|
127
|
+
}
|
|
128
|
+
|
|
99
129
|
console.log("\n========== SYSTEM PERFORMANCE ==========");
|
|
100
|
-
console.log(`CPU: ${cpus()[0]?.model ?? "Unknown"} | RAM: ${(totalmem() / 1024 ** 3).toFixed(2)} GB | Mem: ${bandwidth.toFixed(2)} GB/s | Mem MT: ${multiBandwidth.toFixed(2)} GB/s | Disk: ${storage.toFixed(2)} GB/s`);
|
|
130
|
+
console.log(`CPU: ${cpus()[0]?.model ?? "Unknown"} | RAM: ${(totalmem() / 1024 ** 3).toFixed(2)} GB${gpuInfo} | Mem: ${bandwidth.toFixed(2)} GB/s | Mem MT: ${multiBandwidth.toFixed(2)} GB/s | Disk: ${storage.toFixed(2)} GB/s`);
|
|
101
131
|
console.log("========================================");
|
|
102
132
|
console.log(`${python ? "✅" : "❌"} Python installed`);
|
|
103
|
-
console.log(`${
|
|
133
|
+
console.log(`${backendInstalled ? "✅" : "❌"} ${backendName} installed`);
|
|
104
134
|
|
|
105
|
-
if (!python || !
|
|
106
|
-
console.error(
|
|
135
|
+
if (!python || !backendInstalled) {
|
|
136
|
+
console.error(`\nPython 3 and ${backendName} are required to run model benchmarks.`);
|
|
107
137
|
rl.close();
|
|
108
138
|
process.exit(1);
|
|
109
139
|
}
|
|
110
140
|
|
|
111
141
|
const modelSize = name => {
|
|
112
142
|
if (/-vl\b|-vision\b|whisper/i.test(name)) return 0;
|
|
143
|
+
if (!isMac && /-(?:gguf|awq|gptq|exl2)\b/i.test(name)) return 0;
|
|
113
144
|
const p = name.match(/(\d+(?:\.\d+)?)\s*([bm])(?:-|$)/i);
|
|
114
145
|
const b = name.match(/(?:-|\b)(\d+)-?bit/i);
|
|
115
146
|
if (!p) return 0;
|
|
@@ -127,7 +158,9 @@ const confirmDiskSpace = async (directory, requiredBytes) => {
|
|
|
127
158
|
};
|
|
128
159
|
|
|
129
160
|
const localModels = [];
|
|
161
|
+
const modelDownloadDir = isMac ? "/tmp/mlx-models" : "/tmp/llm-models";
|
|
130
162
|
const roots = [
|
|
163
|
+
modelDownloadDir,
|
|
131
164
|
"/tmp/mlx-models",
|
|
132
165
|
`${homedir()}/.cache/huggingface/hub`,
|
|
133
166
|
`${homedir()}/models`,
|
|
@@ -152,10 +185,13 @@ const scan = async (directory, depth = 0) => {
|
|
|
152
185
|
};
|
|
153
186
|
for (const root of roots) await scan(root);
|
|
154
187
|
|
|
155
|
-
const budget = totalmem() * 0.65;
|
|
188
|
+
const budget = (gpuMem > 0 ? gpuMem : totalmem()) * 0.65;
|
|
156
189
|
const candidates = localModels.filter(m => m.bytes <= budget);
|
|
190
|
+
const hfApiUrl = isMac
|
|
191
|
+
? "https://huggingface.co/api/models?author=mlx-community&search=Instruct&sort=downloads&direction=-1&limit=100"
|
|
192
|
+
: "https://huggingface.co/api/models?pipeline_tag=text-generation&search=Instruct&sort=downloads&direction=-1&limit=100";
|
|
157
193
|
try {
|
|
158
|
-
const response = await fetch(
|
|
194
|
+
const response = await fetch(hfApiUrl, { signal: AbortSignal.timeout(8000) });
|
|
159
195
|
if (response.ok) {
|
|
160
196
|
for (const model of await response.json()) {
|
|
161
197
|
const bytes = modelSize(model.id);
|
|
@@ -177,7 +213,7 @@ for (;;) {
|
|
|
177
213
|
|
|
178
214
|
let selectedModel = model.local ? model.path : null;
|
|
179
215
|
if (!model.local) {
|
|
180
|
-
const modelPath =
|
|
216
|
+
const modelPath = `${modelDownloadDir}/${model.name.replaceAll("/", "--")}`;
|
|
181
217
|
if (await confirmDiskSpace("/tmp", model.bytes * 1.1)) {
|
|
182
218
|
const folderSize = async dir => {
|
|
183
219
|
let bytes = 0;
|
|
@@ -214,7 +250,7 @@ for (;;) {
|
|
|
214
250
|
}
|
|
215
251
|
if (!selectedModel) break;
|
|
216
252
|
|
|
217
|
-
const benchmark = `
|
|
253
|
+
const benchmark = isMac ? `
|
|
218
254
|
import json, os, time
|
|
219
255
|
from mlx_lm import load, stream_generate
|
|
220
256
|
|
|
@@ -235,8 +271,61 @@ for prompt in prompts:
|
|
|
235
271
|
if last_resp and last_resp.generation_tokens:
|
|
236
272
|
ttft = (first - start) if first else 0
|
|
237
273
|
print(json.dumps({"ttft": ttft, "tps": last_resp.generation_tps, "tokens": last_resp.generation_tokens}), flush=True)
|
|
274
|
+
` : `
|
|
275
|
+
import json, os, time, warnings
|
|
276
|
+
warnings.filterwarnings("ignore")
|
|
277
|
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
|
278
|
+
import torch
|
|
279
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
280
|
+
from transformers.generation.streamers import BaseStreamer
|
|
281
|
+
|
|
282
|
+
model_path = os.environ["MODEL_PATH"]
|
|
283
|
+
device = "cuda" if torch.cuda.is_available() else "cpu"
|
|
284
|
+
dtype = torch.float16 if torch.cuda.is_available() else torch.float32
|
|
285
|
+
|
|
286
|
+
tokenizer = AutoTokenizer.from_pretrained(model_path)
|
|
287
|
+
if tokenizer.pad_token_id is None:
|
|
288
|
+
tokenizer.pad_token_id = tokenizer.eos_token_id
|
|
289
|
+
model = AutoModelForCausalLM.from_pretrained(model_path, torch_dtype=dtype, low_cpu_mem_usage=True).to(device)
|
|
290
|
+
|
|
291
|
+
class MetricStreamer(BaseStreamer):
|
|
292
|
+
def __init__(self, start):
|
|
293
|
+
self.start = start
|
|
294
|
+
self.first = None
|
|
295
|
+
self.last = start
|
|
296
|
+
self.count = 0
|
|
297
|
+
self.is_prompt = True
|
|
298
|
+
def put(self, value):
|
|
299
|
+
if self.is_prompt:
|
|
300
|
+
self.is_prompt = False
|
|
301
|
+
return
|
|
302
|
+
now = time.perf_counter()
|
|
303
|
+
if self.first is None: self.first = now
|
|
304
|
+
self.count += 1
|
|
305
|
+
self.last = now
|
|
306
|
+
def end(self): pass
|
|
307
|
+
|
|
308
|
+
prompts = [
|
|
309
|
+
"Explain what RAM does in about 50 words.",
|
|
310
|
+
"Explain why the sky appears blue in about 50 words.",
|
|
311
|
+
"Give a simple three-step recipe for making tea.",
|
|
312
|
+
"Explain the difference between a CPU and a GPU in about 50 words.",
|
|
313
|
+
"Describe the water cycle in about 50 words."
|
|
314
|
+
]
|
|
315
|
+
for prompt in prompts:
|
|
316
|
+
inputs = tokenizer(prompt, return_tensors="pt").to(device)
|
|
317
|
+
start = time.perf_counter()
|
|
318
|
+
streamer = MetricStreamer(start)
|
|
319
|
+
model.generate(**inputs, streamer=streamer, max_new_tokens=128)
|
|
320
|
+
if streamer.count:
|
|
321
|
+
ttft = (streamer.first - start) if streamer.first else 0
|
|
322
|
+
gen_time = (streamer.last - streamer.first) if (streamer.first and streamer.last > streamer.first) else (streamer.last - start)
|
|
323
|
+
tps = streamer.count / gen_time if gen_time > 0 else 0
|
|
324
|
+
print(json.dumps({"ttft": ttft, "tps": tps, "tokens": streamer.count}), flush=True)
|
|
238
325
|
`;
|
|
239
|
-
|
|
326
|
+
|
|
327
|
+
const engineName = isMac ? "MLX" : "Transformers";
|
|
328
|
+
console.log(`\nRunning ${engineName} benchmark for ${model.name}...`);
|
|
240
329
|
const test = spawnProc(python, ["-c", benchmark], {
|
|
241
330
|
env: { ...process.env, MODEL_PATH: selectedModel },
|
|
242
331
|
capture: true
|
|
@@ -244,15 +333,20 @@ for prompt in prompts:
|
|
|
244
333
|
const benchmarkSuccess = (await test.exited) === 0;
|
|
245
334
|
const output = test.getOutput();
|
|
246
335
|
if (benchmarkSuccess) {
|
|
247
|
-
const results = output.trim().split("\n").
|
|
336
|
+
const results = output.trim().split("\n").map(l => {
|
|
337
|
+
try { return JSON.parse(l.trim()); } catch { return null; }
|
|
338
|
+
}).filter(Boolean);
|
|
248
339
|
if (results.length) {
|
|
249
340
|
const meanTps = results.reduce((s, r) => s + r.tps, 0) / results.length;
|
|
250
|
-
console.log(
|
|
341
|
+
console.log(`\n========== ${engineName.toUpperCase()} TOKEN TEST ==========`);
|
|
251
342
|
results.forEach((r, i) => console.log(`${i + 1}. TTFT: ${(r.ttft * 1000).toFixed(0)} ms | Tok/s: ${r.tps.toFixed(2)} | Tokens: ${r.tokens}`));
|
|
252
343
|
console.log(`Mean TTFT: ${(results.reduce((s, r) => s + r.ttft, 0) / results.length * 1000).toFixed(0)} ms | Mean Tok/s: ${meanTps.toFixed(2)}`);
|
|
253
344
|
console.log("=====================================");
|
|
254
345
|
if (/^y(es)?$/i.test(await ask(`Show how fast ${meanTps.toFixed(1)} tok/s looks like? [y/N] `) ?? "")) {
|
|
255
|
-
const
|
|
346
|
+
const demoText = isMac
|
|
347
|
+
? "Artificial intelligence and large language models process text by predicting the next token in a sequence. With Apple Silicon unified memory architecture, weights are streamed with high bandwidth directly to the GPU cores. "
|
|
348
|
+
: "Artificial intelligence and large language models process text by predicting the next token in a sequence. Weights and activations are streamed with high bandwidth directly to the compute cores for fast token generation. ";
|
|
349
|
+
const words = demoText.repeat(5).match(/\s*\S+/g);
|
|
256
350
|
const demoStart = performance.now();
|
|
257
351
|
for (let i = 0; performance.now() - demoStart < 4000; i++) {
|
|
258
352
|
process.stdout.write(words[i % words.length]);
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "llm-sizer",
|
|
3
|
-
"version": "0.1.
|
|
4
|
-
"description": "Benchmark
|
|
3
|
+
"version": "0.1.1",
|
|
4
|
+
"description": "Benchmark memory bandwidth, disk speed, and LLM inference performance (MLX on macOS, Transformers on Linux) to find the optimal model size",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "llm-sizer.js",
|
|
7
7
|
"bin": {
|
|
@@ -17,10 +17,12 @@
|
|
|
17
17
|
},
|
|
18
18
|
"keywords": [
|
|
19
19
|
"mlx",
|
|
20
|
+
"transformers",
|
|
20
21
|
"benchmark",
|
|
21
22
|
"apple-silicon",
|
|
22
23
|
"llm",
|
|
23
24
|
"mac",
|
|
25
|
+
"linux",
|
|
24
26
|
"inference"
|
|
25
27
|
],
|
|
26
28
|
"license": "MIT"
|