pichak 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ __pycache__/
2
+ *.pyc
3
+ dist/
4
+ build/
5
+ *.egg-info/
6
+ .pytest_cache/
pichak-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Oleksandr Pichak
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
pichak-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,190 @@
1
+ Metadata-Version: 2.5
2
+ Name: pichak
3
+ Version: 0.1.0
4
+ Summary: Derive fine-tuning hyperparameters from measurements of your actual machine, and print the measurement behind every number.
5
+ Project-URL: Homepage, https://github.com/olesxg/pichak
6
+ Project-URL: Findings, https://github.com/olesxg/flap-findings
7
+ Author-email: Oleksandr Pichak <stasselust@gmail.com>
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: batch-size,benchmark,fine-tuning,gpu,hyperparameters,learning-rate,llm,lora,vram
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Requires-Python: >=3.9
18
+ Provides-Extra: dev
19
+ Requires-Dist: pytest>=7; extra == 'dev'
20
+ Provides-Extra: hf
21
+ Requires-Dist: safetensors>=0.4; extra == 'hf'
22
+ Requires-Dist: torch>=2.0; extra == 'hf'
23
+ Requires-Dist: transformers>=4.40; extra == 'hf'
24
+ Provides-Extra: torch
25
+ Requires-Dist: torch>=2.0; extra == 'torch'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # pichak
29
+
30
+ **Fine-tuning hyperparameters derived from your machine, with the measurement
31
+ printed next to every number.**
32
+
33
+ ```bash
34
+ pip install pichak
35
+ ```
36
+
37
+ ```python
38
+ from pichak import derive
39
+
40
+ plan = derive(
41
+ data="train.jsonl",
42
+ tokenizer=tok,
43
+ step_fn=lambda b: one_real_forward_and_backward(b),
44
+ named_weights=model.named_parameters(),
45
+ )
46
+ print(plan.report())
47
+ ```
48
+
49
+ ```
50
+ seq_len 512
51
+ token p95 416 over 512 real rows, rounded up to 512; p99 is
52
+ 657 and max 943, so 5% of rows lose their tail at this length
53
+
54
+ micro_batch 3
55
+ a doubling ramp with a real forward+backward at each rung:
56
+ 1=ok, 2=ok, 4=ok, 8=SPILL. 8 SPILLED to system memory — it did
57
+ not raise, it got 1.6x slower per sample, which is the Windows
58
+ failure mode, so 4 is the largest that fit, peaking at 5.05GB
59
+ of 5.05GB; kept 3 to leave 12% for optimizer state
60
+
61
+ learning_rate 0.000282
62
+ 1/32 of the scale at which one Adam step would rewrite rather
63
+ than perturb. ||W||_F/sqrt(numel) over 26 matrices; the
64
+ smallest binds and it is b.1.f.2.weight at 0.00902
65
+
66
+ NOT MEASURED (1): target_tokens=65536
67
+ ```
68
+
69
+ ---
70
+
71
+ ## Why not just use a good default
72
+
73
+ Because you cannot tell a good default from a bad one when it fails.
74
+
75
+ `micro_batch=4` tells you nothing about whether 4 came from a measurement on your
76
+ card, a heuristic in a blog post, or a number someone picked in 2023 for a
77
+ different model. So when it OOMs at step 40 you bisect instead of reading.
78
+
79
+ Every number `pichak` returns carries the sentence that produced it. Anything it
80
+ could **not** measure is collected under `plan.constants()` and printed together,
81
+ so a constant can never quietly pass for a derivation.
82
+
83
+ ```python
84
+ plan.micro_batch # 3
85
+ plan.why("micro_batch") # the ramp, rung by rung, and which one failed
86
+ plan.constants() # {'target_tokens': 65536}
87
+ ```
88
+
89
+ ## It catches the failure that does not raise
90
+
91
+ On Windows, a batch that exceeds VRAM does not throw. The driver pages the excess
92
+ to system memory and the step simply crawls. Measured on a GTX 1060 6GB with a
93
+ 6-layer model at sequence 512:
94
+
95
+ | | micro_batch | peak | seconds/step |
96
+ |---|---|---|---|
97
+ | watching only for OOM | 14 | **18.24GB** on a 6GB card | 47.5 |
98
+ | watching per-sample time too | 3 | 5.05GB | **2.0** |
99
+
100
+ Both "work". The first is 23x slower and nothing in the logs says why. On a healthy
101
+ ramp, seconds-per-sample is flat or improves as the batch grows — bigger batches
102
+ amortise fixed costs. A sudden jump means the batch did not really fit.
103
+
104
+ ## What it derives, and from what
105
+
106
+ | | measurement |
107
+ |---|---|
108
+ | `seq_len` | tokenised p95 over 512 real rows of your corpus |
109
+ | `loss_policy` | whether the rows have a separable completion to mask against |
110
+ | `micro_batch` | a doubling ramp with a real forward+backward at each rung |
111
+ | `seconds_per_step` | the wall time of that rung |
112
+ | `grad_accum` | micro-batch x seq_len against your token target |
113
+ | `lora_rank` | largest rung whose optimizer state fits — fp32 master, two Adam moments, and one step's gradients, which is the copy count people forget |
114
+ | `learning_rate` | the model's own Frobenius norms: the step size at which one Adam update would rewrite a weight rather than perturb it |
115
+
116
+ Everything is optional. Pass only a corpus and a tokenizer and you get a sequence
117
+ length; pass a `step_fn` and you get the batch. It will not invent a number it
118
+ could not measure — a plan with three derived values and an honest gap is more
119
+ useful than one with ten you cannot tell apart.
120
+
121
+ ## The learning rate, since it is the surprising one
122
+
123
+ An Adam update moves a parameter by roughly `lr`, whatever the gradient's scale. So
124
+ there is a learning rate at which one step rewrites a weight matrix instead of
125
+ nudging it: where `lr * sqrt(numel(W))` reaches `||W||_F`. Every sane learning rate
126
+ is a fraction of that, and the fraction is what schedules argue about.
127
+
128
+ Measured, at the ranks behind this library:
129
+
130
+ ```
131
+ rewrite/2 2.26e-3 destroyed the model twice (CE 1.74 -> 10.4, 1.11 -> 19.3)
132
+ rewrite/32 1.41e-4 the first scale that actually learned
133
+ rewrite/45 where a hand-tuned LoRA run at 2e-4 landed
134
+ ```
135
+
136
+ `pichak` opens at `rewrite/32` — the top of the range that works. Opening hot costs
137
+ the model; opening cold costs time. With a penalty that asymmetric you start at the
138
+ bottom of what is known.
139
+
140
+ This is not tuned for your dataset. It is the scale at which updates are the right
141
+ *size* for these weights. It replaces "I copied 2e-4 from a post about a different
142
+ model", which is the actual alternative.
143
+
144
+ ## Measurement tools, usable on their own
145
+
146
+ ```bash
147
+ pichak gpu # virtualisation, launch latency, transfer bandwidth
148
+ pichak disk "models/*.safetensors" # queue-depth sweep, page cache bypassed
149
+ pichak corpus train.jsonl mistralai/Mistral-Small-24B-Instruct-2501
150
+ ```
151
+
152
+ ```python
153
+ from pichak.measure import disk_queue_depth, transfer_bandwidth, virtualisation
154
+ ```
155
+
156
+ These exist because the numbers people quote are rarely the numbers their machine
157
+ gives:
158
+
159
+ - A buffered disk benchmark reported **4079 MB/s on a drive rated 2100** — that was
160
+ the page cache. Unbuffered, the same drive peaks at **queue depth 3** and gets
161
+ *slower* past it.
162
+ - A rented A6000 measured **D2H 457 MB/s against H2D 1502**, and pinned memory —
163
+ the standard fix — bought nothing. Nothing about PCIe explains that; it was a
164
+ fabric, and it decided where hidden states could live.
165
+ - One rented machine had **84.8us kernel launches** against a normal 2-5. A
166
+ workload issuing 400,000 launches per step would have spent nine hours on
167
+ overhead and looked slower than a 2016 card.
168
+
169
+ ## Install
170
+
171
+ ```bash
172
+ pip install pichak # the plan, the disk sweep — no dependencies
173
+ pip install pichak[torch] # the ramp, the learning rate, the GPU measurements
174
+ pip install pichak[hf] # + transformers, for the corpus CLI
175
+ ```
176
+
177
+ Python 3.9+. The core has **no dependencies at all**; torch is only needed for the
178
+ parts that touch a GPU.
179
+
180
+ ## Where the numbers come from
181
+
182
+ Every measurement quoted here was made while fine-tuning a 24B model on a 6GB
183
+ GTX 1060 and on a rented A6000 held down to the same 6GB. The raw logs — including
184
+ the six runs that failed before one worked — are at
185
+ [flap-findings](https://github.com/olesxg/flap-findings), along with the five
186
+ hypotheses that sounded right and were wrong.
187
+
188
+ ## Licence
189
+
190
+ MIT. By [Oleksandr Pichak](https://github.com/olesxg).
pichak-0.1.0/README.md ADDED
@@ -0,0 +1,163 @@
1
+ # pichak
2
+
3
+ **Fine-tuning hyperparameters derived from your machine, with the measurement
4
+ printed next to every number.**
5
+
6
+ ```bash
7
+ pip install pichak
8
+ ```
9
+
10
+ ```python
11
+ from pichak import derive
12
+
13
+ plan = derive(
14
+ data="train.jsonl",
15
+ tokenizer=tok,
16
+ step_fn=lambda b: one_real_forward_and_backward(b),
17
+ named_weights=model.named_parameters(),
18
+ )
19
+ print(plan.report())
20
+ ```
21
+
22
+ ```
23
+ seq_len 512
24
+ token p95 416 over 512 real rows, rounded up to 512; p99 is
25
+ 657 and max 943, so 5% of rows lose their tail at this length
26
+
27
+ micro_batch 3
28
+ a doubling ramp with a real forward+backward at each rung:
29
+ 1=ok, 2=ok, 4=ok, 8=SPILL. 8 SPILLED to system memory — it did
30
+ not raise, it got 1.6x slower per sample, which is the Windows
31
+ failure mode, so 4 is the largest that fit, peaking at 5.05GB
32
+ of 5.05GB; kept 3 to leave 12% for optimizer state
33
+
34
+ learning_rate 0.000282
35
+ 1/32 of the scale at which one Adam step would rewrite rather
36
+ than perturb. ||W||_F/sqrt(numel) over 26 matrices; the
37
+ smallest binds and it is b.1.f.2.weight at 0.00902
38
+
39
+ NOT MEASURED (1): target_tokens=65536
40
+ ```
41
+
42
+ ---
43
+
44
+ ## Why not just use a good default
45
+
46
+ Because you cannot tell a good default from a bad one when it fails.
47
+
48
+ `micro_batch=4` tells you nothing about whether 4 came from a measurement on your
49
+ card, a heuristic in a blog post, or a number someone picked in 2023 for a
50
+ different model. So when it OOMs at step 40 you bisect instead of reading.
51
+
52
+ Every number `pichak` returns carries the sentence that produced it. Anything it
53
+ could **not** measure is collected under `plan.constants()` and printed together,
54
+ so a constant can never quietly pass for a derivation.
55
+
56
+ ```python
57
+ plan.micro_batch # 3
58
+ plan.why("micro_batch") # the ramp, rung by rung, and which one failed
59
+ plan.constants() # {'target_tokens': 65536}
60
+ ```
61
+
62
+ ## It catches the failure that does not raise
63
+
64
+ On Windows, a batch that exceeds VRAM does not throw. The driver pages the excess
65
+ to system memory and the step simply crawls. Measured on a GTX 1060 6GB with a
66
+ 6-layer model at sequence 512:
67
+
68
+ | | micro_batch | peak | seconds/step |
69
+ |---|---|---|---|
70
+ | watching only for OOM | 14 | **18.24GB** on a 6GB card | 47.5 |
71
+ | watching per-sample time too | 3 | 5.05GB | **2.0** |
72
+
73
+ Both "work". The first is 23x slower and nothing in the logs says why. On a healthy
74
+ ramp, seconds-per-sample is flat or improves as the batch grows — bigger batches
75
+ amortise fixed costs. A sudden jump means the batch did not really fit.
76
+
77
+ ## What it derives, and from what
78
+
79
+ | | measurement |
80
+ |---|---|
81
+ | `seq_len` | tokenised p95 over 512 real rows of your corpus |
82
+ | `loss_policy` | whether the rows have a separable completion to mask against |
83
+ | `micro_batch` | a doubling ramp with a real forward+backward at each rung |
84
+ | `seconds_per_step` | the wall time of that rung |
85
+ | `grad_accum` | micro-batch x seq_len against your token target |
86
+ | `lora_rank` | largest rung whose optimizer state fits — fp32 master, two Adam moments, and one step's gradients, which is the copy count people forget |
87
+ | `learning_rate` | the model's own Frobenius norms: the step size at which one Adam update would rewrite a weight rather than perturb it |
88
+
89
+ Everything is optional. Pass only a corpus and a tokenizer and you get a sequence
90
+ length; pass a `step_fn` and you get the batch. It will not invent a number it
91
+ could not measure — a plan with three derived values and an honest gap is more
92
+ useful than one with ten you cannot tell apart.
93
+
94
+ ## The learning rate, since it is the surprising one
95
+
96
+ An Adam update moves a parameter by roughly `lr`, whatever the gradient's scale. So
97
+ there is a learning rate at which one step rewrites a weight matrix instead of
98
+ nudging it: where `lr * sqrt(numel(W))` reaches `||W||_F`. Every sane learning rate
99
+ is a fraction of that, and the fraction is what schedules argue about.
100
+
101
+ Measured, at the ranks behind this library:
102
+
103
+ ```
104
+ rewrite/2 2.26e-3 destroyed the model twice (CE 1.74 -> 10.4, 1.11 -> 19.3)
105
+ rewrite/32 1.41e-4 the first scale that actually learned
106
+ rewrite/45 where a hand-tuned LoRA run at 2e-4 landed
107
+ ```
108
+
109
+ `pichak` opens at `rewrite/32` — the top of the range that works. Opening hot costs
110
+ the model; opening cold costs time. With a penalty that asymmetric you start at the
111
+ bottom of what is known.
112
+
113
+ This is not tuned for your dataset. It is the scale at which updates are the right
114
+ *size* for these weights. It replaces "I copied 2e-4 from a post about a different
115
+ model", which is the actual alternative.
116
+
117
+ ## Measurement tools, usable on their own
118
+
119
+ ```bash
120
+ pichak gpu # virtualisation, launch latency, transfer bandwidth
121
+ pichak disk "models/*.safetensors" # queue-depth sweep, page cache bypassed
122
+ pichak corpus train.jsonl mistralai/Mistral-Small-24B-Instruct-2501
123
+ ```
124
+
125
+ ```python
126
+ from pichak.measure import disk_queue_depth, transfer_bandwidth, virtualisation
127
+ ```
128
+
129
+ These exist because the numbers people quote are rarely the numbers their machine
130
+ gives:
131
+
132
+ - A buffered disk benchmark reported **4079 MB/s on a drive rated 2100** — that was
133
+ the page cache. Unbuffered, the same drive peaks at **queue depth 3** and gets
134
+ *slower* past it.
135
+ - A rented A6000 measured **D2H 457 MB/s against H2D 1502**, and pinned memory —
136
+ the standard fix — bought nothing. Nothing about PCIe explains that; it was a
137
+ fabric, and it decided where hidden states could live.
138
+ - One rented machine had **84.8us kernel launches** against a normal 2-5. A
139
+ workload issuing 400,000 launches per step would have spent nine hours on
140
+ overhead and looked slower than a 2016 card.
141
+
142
+ ## Install
143
+
144
+ ```bash
145
+ pip install pichak # the plan, the disk sweep — no dependencies
146
+ pip install pichak[torch] # the ramp, the learning rate, the GPU measurements
147
+ pip install pichak[hf] # + transformers, for the corpus CLI
148
+ ```
149
+
150
+ Python 3.9+. The core has **no dependencies at all**; torch is only needed for the
151
+ parts that touch a GPU.
152
+
153
+ ## Where the numbers come from
154
+
155
+ Every measurement quoted here was made while fine-tuning a 24B model on a 6GB
156
+ GTX 1060 and on a rented A6000 held down to the same 6GB. The raw logs — including
157
+ the six runs that failed before one worked — are at
158
+ [flap-findings](https://github.com/olesxg/flap-findings), along with the five
159
+ hypotheses that sounded right and were wrong.
160
+
161
+ ## Licence
162
+
163
+ MIT. By [Oleksandr Pichak](https://github.com/olesxg).
@@ -0,0 +1,29 @@
1
+ """pichak — derive fine-tuning hyperparameters from your machine, not from a blog.
2
+
3
+ from pichak import derive
4
+ plan = derive(data="train.jsonl", tokenizer=tok,
5
+ step_fn=lambda b: one_real_step(b),
6
+ named_weights=model.named_parameters())
7
+ print(plan.report())
8
+
9
+ Every number comes back with the measurement that produced it, and anything NOT
10
+ measured is listed together under `plan.constants()` so a constant can never pass
11
+ for a derivation.
12
+
13
+ By Oleksandr Pichak. The measurements this is built on, including the wrong turns,
14
+ are at https://github.com/olesxg/flap-findings
15
+ """
16
+
17
+ from .plan import CONSTANT, Plan, Value
18
+ from .tune import (adapter_bytes_per_rank, auto_rank, derive, derive_lr,
19
+ grad_accum_for, profile_corpus, ramp_micro_batch,
20
+ rewrite_scale)
21
+
22
+ __version__ = "0.1.0"
23
+ __author__ = "Oleksandr Pichak"
24
+ __all__ = [
25
+ "derive", "Plan", "Value", "CONSTANT",
26
+ "profile_corpus", "ramp_micro_batch", "auto_rank", "adapter_bytes_per_rank",
27
+ "grad_accum_for", "derive_lr", "rewrite_scale",
28
+ "__version__", "__author__",
29
+ ]
@@ -0,0 +1,64 @@
1
+ """`pichak` on the command line — measure this machine, print what it gives.
2
+
3
+ pichak gpu virtualisation, launch latency, transfer bandwidth
4
+ pichak disk <glob> queue-depth sweep with the page cache bypassed
5
+ pichak corpus <jsonl> <tokenizer> sequence length from real rows
6
+ """
7
+ import sys
8
+
9
+
10
+ def main(argv=None) -> int:
11
+ argv = list(sys.argv[1:] if argv is None else argv)
12
+ cmd = argv[0] if argv else "help"
13
+
14
+ if cmd == "gpu":
15
+ from .measure import kernel_launch_us, transfer_bandwidth, virtualisation
16
+ v = virtualisation()
17
+ print(f"gpu : {v['gpu']}")
18
+ print(f"virt mode : {v['virtualisation_mode']}")
19
+ print(f"host : {v['host']} ({v['note']})")
20
+ if v["virtualised"]:
21
+ print("\nVIRTUALISED — every timing below is the hypervisor's, not "
22
+ "the card's.")
23
+ try:
24
+ import torch
25
+ except ImportError:
26
+ print("\ntorch is not installed; install pichak[torch] for the rest.")
27
+ return 2
28
+ if not torch.cuda.is_available():
29
+ print("\nno CUDA visible.")
30
+ return 2
31
+ us = kernel_launch_us(torch)
32
+ print(f"\nkernel launch: {us:.1f} us "
33
+ f"(2-5 normal passthrough; 85 seen on a network-attached vGPU)")
34
+ print(" judge this as a SHARE of your step, never in raw microseconds")
35
+ print()
36
+ print(transfer_bandwidth(torch).describe())
37
+ return 0
38
+
39
+ if cmd == "disk":
40
+ if len(argv) < 2:
41
+ print("usage: pichak disk <glob of several large files>")
42
+ return 2
43
+ from .measure import disk_queue_depth
44
+ disk_queue_depth(argv[1:])
45
+ return 0
46
+
47
+ if cmd == "corpus":
48
+ if len(argv) < 3:
49
+ print("usage: pichak corpus <train.jsonl> <tokenizer-name-or-path>")
50
+ return 2
51
+ from transformers import AutoTokenizer
52
+ from .tune import profile_corpus
53
+ tok = AutoTokenizer.from_pretrained(argv[2])
54
+ prof = profile_corpus(argv[1], tok)
55
+ print(prof.describe())
56
+ print(f"\nseq_len {prof.seq_len}\n {prof.seq_source}")
57
+ return 0
58
+
59
+ print(__doc__)
60
+ return 0 if cmd in ("help", "-h", "--help") else 2
61
+
62
+
63
+ if __name__ == "__main__":
64
+ raise SystemExit(main())
@@ -0,0 +1,15 @@
1
+ """Measurements that answer one question each, with no training loop attached.
2
+
3
+ from pichak.measure import disk_queue_depth, transfer_bandwidth
4
+
5
+ These exist because the numbers people quote for storage and PCIe are almost never
6
+ the numbers their machine gives. A buffered disk benchmark on a host with free RAM
7
+ reports the page cache; a rented GPU can have a 3.4x asymmetry between its transfer
8
+ directions that no datasheet mentions.
9
+ """
10
+
11
+ from .disk import disk_queue_depth
12
+ from .gpu import kernel_launch_us, transfer_bandwidth, virtualisation
13
+
14
+ __all__ = ["disk_queue_depth", "kernel_launch_us", "transfer_bandwidth",
15
+ "virtualisation"]
@@ -0,0 +1,144 @@
1
+ """How many concurrent readers does this drive actually want?
2
+
3
+ WHY YOU CANNOT USE A BUFFERED BENCHMARK. The first version of this measurement,
4
+ run with ordinary `open()`, reported:
5
+
6
+ QD1 1158 MB/s QD4 781 MB/s <- slower than one reader
7
+ QD8 4079 MB/s <- above the drive's rated 2100
8
+
9
+ Those are the OS page cache answering, not the disk. If a storage benchmark tells
10
+ you more than the hardware can do, it has stopped measuring the hardware. On
11
+ Windows this bypasses it with `FILE_FLAG_NO_BUFFERING`, which is why the reads have
12
+ to be sector-aligned and why this goes through ctypes.
13
+
14
+ WHAT IT FOUND. On a DRAM-less Kingston NV1:
15
+
16
+ QD1 1017 MB/s QD2 1655 QD3 1703 (optimum) QD4 1535 QD8 1035
17
+
18
+ More threads is not more throughput. A DRAM-less controller degrades past its sweet
19
+ spot rather than saturating, so past QD3 it simply loses — and eight readers, the
20
+ number a reasonable person would pick, is no better than one.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import ctypes
26
+ import glob
27
+ import os
28
+ import sys
29
+ import time
30
+ from concurrent.futures import ThreadPoolExecutor
31
+ from typing import Dict, List, Optional, Sequence
32
+
33
+ MB = 1024 ** 2
34
+ SECTOR = 4096
35
+
36
+
37
+ def _unbuffered_reader():
38
+ """(reader, available) — an unbuffered read on Windows, else plain reads."""
39
+ if not sys.platform.startswith("win"):
40
+ def posix(path: str, off: int, nbytes: int, block: int) -> int:
41
+ # O_DIRECT needs alignment and is not portable across filesystems;
42
+ # POSIX_FADV_DONTNEED after the read is the honest compromise.
43
+ fd = os.open(path, os.O_RDONLY)
44
+ got = 0
45
+ try:
46
+ os.lseek(fd, off, os.SEEK_SET)
47
+ while got < nbytes:
48
+ b = os.read(fd, min(block, nbytes - got))
49
+ if not b:
50
+ break
51
+ got += len(b)
52
+ if hasattr(os, "posix_fadvise"):
53
+ os.posix_fadvise(fd, off, nbytes, os.POSIX_FADV_DONTNEED)
54
+ finally:
55
+ os.close(fd)
56
+ return got
57
+ return posix, False
58
+
59
+ k32 = ctypes.WinDLL("kernel32", use_last_error=True)
60
+ from ctypes import wintypes
61
+ GENERIC_READ, FILE_SHARE_READ, OPEN_EXISTING = 0x80000000, 1, 3
62
+ NO_BUFFERING, SEQUENTIAL = 0x20000000, 0x08000000
63
+ MEM_COMMIT_RESERVE, PAGE_RW, MEM_RELEASE = 0x3000, 4, 0x8000
64
+ k32.CreateFileW.restype = wintypes.HANDLE
65
+ k32.CreateFileW.argtypes = [wintypes.LPCWSTR, wintypes.DWORD, wintypes.DWORD,
66
+ ctypes.c_void_p, wintypes.DWORD, wintypes.DWORD,
67
+ wintypes.HANDLE]
68
+ k32.VirtualAlloc.restype = ctypes.c_void_p
69
+ k32.VirtualAlloc.argtypes = [ctypes.c_void_p, ctypes.c_size_t,
70
+ wintypes.DWORD, wintypes.DWORD]
71
+ k32.ReadFile.argtypes = [wintypes.HANDLE, ctypes.c_void_p, wintypes.DWORD,
72
+ ctypes.POINTER(wintypes.DWORD), ctypes.c_void_p]
73
+ k32.SetFilePointerEx.argtypes = [wintypes.HANDLE, ctypes.c_longlong,
74
+ ctypes.c_void_p, wintypes.DWORD]
75
+
76
+ def win(path: str, off: int, nbytes: int, block: int) -> int:
77
+ h = k32.CreateFileW(path, GENERIC_READ, FILE_SHARE_READ, None,
78
+ OPEN_EXISTING, NO_BUFFERING | SEQUENTIAL, None)
79
+ if h == wintypes.HANDLE(-1).value:
80
+ raise OSError(ctypes.get_last_error())
81
+ buf = k32.VirtualAlloc(None, block, MEM_COMMIT_RESERVE, PAGE_RW)
82
+ got, done = 0, wintypes.DWORD(0)
83
+ try:
84
+ k32.SetFilePointerEx(h, ctypes.c_longlong(off), None, 0)
85
+ while got < nbytes:
86
+ if not k32.ReadFile(h, ctypes.c_void_p(buf), block,
87
+ ctypes.byref(done), None):
88
+ raise OSError(ctypes.get_last_error())
89
+ if done.value == 0:
90
+ break
91
+ got += done.value
92
+ finally:
93
+ k32.VirtualFree(ctypes.c_void_p(buf), 0, MEM_RELEASE)
94
+ k32.CloseHandle(h)
95
+ return got
96
+ return win, True
97
+
98
+
99
+ def disk_queue_depth(paths: Sequence[str], *, depths: Sequence[int] = (1, 2, 3, 4, 6, 8),
100
+ chunk_mb: int = 256, block_mb: int = 8,
101
+ log=print) -> Dict[int, float]:
102
+ """MB/s at each queue depth. Returns {depth: rate}, and prints as it goes.
103
+
104
+ ``paths`` should be several large files — model shards are ideal. Workers read
105
+ DIFFERENT files at DIFFERENT offsets so that a second reader cannot be served
106
+ from the first one's readahead.
107
+ """
108
+ files = [p for pat in paths for p in sorted(glob.glob(pat))] if any(
109
+ "*" in p or "?" in p for p in paths) else list(paths)
110
+ files = [f for f in files if os.path.isfile(f)]
111
+ if not files:
112
+ raise FileNotFoundError(f"no readable files in {list(paths)!r}")
113
+
114
+ read, unbuffered = _unbuffered_reader()
115
+ if not unbuffered:
116
+ log(" note: unbuffered reads are Windows-only here; on this platform the "
117
+ "page cache is advised away after each read, which is weaker. Treat "
118
+ "any result above the drive's rating as cache.")
119
+
120
+ chunk, block = chunk_mb * MB, block_mb * MB
121
+ out: Dict[int, float] = {}
122
+ log(f" {len(files)} file(s), {chunk_mb}MB per worker, {block_mb}MB blocks")
123
+ for d in depths:
124
+ jobs = []
125
+ for i in range(d):
126
+ p = files[i % len(files)]
127
+ size = (os.path.getsize(p) // SECTOR) * SECTOR
128
+ slot = i // len(files)
129
+ off = ((slot * 1300 * MB + 300 * MB) % max(SECTOR, size - chunk))
130
+ jobs.append((p, (off // SECTOR) * SECTOR, chunk, block))
131
+ t0 = time.perf_counter()
132
+ with ThreadPoolExecutor(max_workers=d) as ex:
133
+ total = sum(ex.map(lambda a: read(*a), jobs))
134
+ rate = total / (time.perf_counter() - t0) / MB
135
+ out[d] = rate
136
+ log(f" QD{d:<3d} {total / MB:6.0f} MB in {(total / MB) / rate:5.2f}s "
137
+ f"= {rate:6.0f} MB/s")
138
+ best = max(out, key=out.get)
139
+ log(f" optimum QD{best} at {out[best]:.0f} MB/s "
140
+ f"({out[best] / out[min(out)]:.2f}x QD{min(out)})")
141
+ if best < max(out):
142
+ log(" more readers than that LOSE — a DRAM-less controller degrades past "
143
+ "its sweet spot rather than saturating.")
144
+ return out