pichak 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pichak-0.1.0/.gitignore +6 -0
- pichak-0.1.0/LICENSE +21 -0
- pichak-0.1.0/PKG-INFO +190 -0
- pichak-0.1.0/README.md +163 -0
- pichak-0.1.0/pichak/__init__.py +29 -0
- pichak-0.1.0/pichak/__main__.py +64 -0
- pichak-0.1.0/pichak/measure/__init__.py +15 -0
- pichak-0.1.0/pichak/measure/disk.py +144 -0
- pichak-0.1.0/pichak/measure/gpu.py +141 -0
- pichak-0.1.0/pichak/plan.py +199 -0
- pichak-0.1.0/pichak/py.typed +0 -0
- pichak-0.1.0/pichak/tune/__init__.py +134 -0
- pichak-0.1.0/pichak/tune/corpus.py +161 -0
- pichak-0.1.0/pichak/tune/lr.py +107 -0
- pichak-0.1.0/pichak/tune/memory.py +222 -0
- pichak-0.1.0/pyproject.toml +38 -0
- pichak-0.1.0/tests/test_corpus.py +74 -0
- pichak-0.1.0/tests/test_memory_and_lr.py +207 -0
- pichak-0.1.0/tests/test_plan.py +77 -0
pichak-0.1.0/.gitignore
ADDED
pichak-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Oleksandr Pichak
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
pichak-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pichak
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Derive fine-tuning hyperparameters from measurements of your actual machine, and print the measurement behind every number.
|
|
5
|
+
Project-URL: Homepage, https://github.com/olesxg/pichak
|
|
6
|
+
Project-URL: Findings, https://github.com/olesxg/flap-findings
|
|
7
|
+
Author-email: Oleksandr Pichak <stasselust@gmail.com>
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: batch-size,benchmark,fine-tuning,gpu,hyperparameters,learning-rate,llm,lora,vram
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Requires-Python: >=3.9
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
20
|
+
Provides-Extra: hf
|
|
21
|
+
Requires-Dist: safetensors>=0.4; extra == 'hf'
|
|
22
|
+
Requires-Dist: torch>=2.0; extra == 'hf'
|
|
23
|
+
Requires-Dist: transformers>=4.40; extra == 'hf'
|
|
24
|
+
Provides-Extra: torch
|
|
25
|
+
Requires-Dist: torch>=2.0; extra == 'torch'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# pichak
|
|
29
|
+
|
|
30
|
+
**Fine-tuning hyperparameters derived from your machine, with the measurement
|
|
31
|
+
printed next to every number.**
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install pichak
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from pichak import derive
|
|
39
|
+
|
|
40
|
+
plan = derive(
|
|
41
|
+
data="train.jsonl",
|
|
42
|
+
tokenizer=tok,
|
|
43
|
+
step_fn=lambda b: one_real_forward_and_backward(b),
|
|
44
|
+
named_weights=model.named_parameters(),
|
|
45
|
+
)
|
|
46
|
+
print(plan.report())
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
seq_len 512
|
|
51
|
+
token p95 416 over 512 real rows, rounded up to 512; p99 is
|
|
52
|
+
657 and max 943, so 5% of rows lose their tail at this length
|
|
53
|
+
|
|
54
|
+
micro_batch 3
|
|
55
|
+
a doubling ramp with a real forward+backward at each rung:
|
|
56
|
+
1=ok, 2=ok, 4=ok, 8=SPILL. 8 SPILLED to system memory — it did
|
|
57
|
+
not raise, it got 1.6x slower per sample, which is the Windows
|
|
58
|
+
failure mode, so 4 is the largest that fit, peaking at 5.05GB
|
|
59
|
+
of 5.05GB; kept 3 to leave 12% for optimizer state
|
|
60
|
+
|
|
61
|
+
learning_rate 0.000282
|
|
62
|
+
1/32 of the scale at which one Adam step would rewrite rather
|
|
63
|
+
than perturb. ||W||_F/sqrt(numel) over 26 matrices; the
|
|
64
|
+
smallest binds and it is b.1.f.2.weight at 0.00902
|
|
65
|
+
|
|
66
|
+
NOT MEASURED (1): target_tokens=65536
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Why not just use a good default
|
|
72
|
+
|
|
73
|
+
Because you cannot tell a good default from a bad one when it fails.
|
|
74
|
+
|
|
75
|
+
`micro_batch=4` tells you nothing about whether 4 came from a measurement on your
|
|
76
|
+
card, a heuristic in a blog post, or a number someone picked in 2023 for a
|
|
77
|
+
different model. So when it OOMs at step 40 you bisect instead of reading.
|
|
78
|
+
|
|
79
|
+
Every number `pichak` returns carries the sentence that produced it. Anything it
|
|
80
|
+
could **not** measure is collected under `plan.constants()` and printed together,
|
|
81
|
+
so a constant can never quietly pass for a derivation.
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
plan.micro_batch # 3
|
|
85
|
+
plan.why("micro_batch") # the ramp, rung by rung, and which one failed
|
|
86
|
+
plan.constants() # {'target_tokens': 65536}
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## It catches the failure that does not raise
|
|
90
|
+
|
|
91
|
+
On Windows, a batch that exceeds VRAM does not throw. The driver pages the excess
|
|
92
|
+
to system memory and the step simply crawls. Measured on a GTX 1060 6GB with a
|
|
93
|
+
6-layer model at sequence 512:
|
|
94
|
+
|
|
95
|
+
| | micro_batch | peak | seconds/step |
|
|
96
|
+
|---|---|---|---|
|
|
97
|
+
| watching only for OOM | 14 | **18.24GB** on a 6GB card | 47.5 |
|
|
98
|
+
| watching per-sample time too | 3 | 5.05GB | **2.0** |
|
|
99
|
+
|
|
100
|
+
Both "work". The first is 23x slower and nothing in the logs says why. On a healthy
|
|
101
|
+
ramp, seconds-per-sample is flat or improves as the batch grows — bigger batches
|
|
102
|
+
amortise fixed costs. A sudden jump means the batch did not really fit.
|
|
103
|
+
|
|
104
|
+
## What it derives, and from what
|
|
105
|
+
|
|
106
|
+
| | measurement |
|
|
107
|
+
|---|---|
|
|
108
|
+
| `seq_len` | tokenised p95 over 512 real rows of your corpus |
|
|
109
|
+
| `loss_policy` | whether the rows have a separable completion to mask against |
|
|
110
|
+
| `micro_batch` | a doubling ramp with a real forward+backward at each rung |
|
|
111
|
+
| `seconds_per_step` | the wall time of that rung |
|
|
112
|
+
| `grad_accum` | micro-batch x seq_len against your token target |
|
|
113
|
+
| `lora_rank` | largest rung whose optimizer state fits — fp32 master, two Adam moments, and one step's gradients, which is the copy count people forget |
|
|
114
|
+
| `learning_rate` | the model's own Frobenius norms: the step size at which one Adam update would rewrite a weight rather than perturb it |
|
|
115
|
+
|
|
116
|
+
Everything is optional. Pass only a corpus and a tokenizer and you get a sequence
|
|
117
|
+
length; pass a `step_fn` and you get the batch. It will not invent a number it
|
|
118
|
+
could not measure — a plan with three derived values and an honest gap is more
|
|
119
|
+
useful than one with ten you cannot tell apart.
|
|
120
|
+
|
|
121
|
+
## The learning rate, since it is the surprising one
|
|
122
|
+
|
|
123
|
+
An Adam update moves a parameter by roughly `lr`, whatever the gradient's scale. So
|
|
124
|
+
there is a learning rate at which one step rewrites a weight matrix instead of
|
|
125
|
+
nudging it: where `lr * sqrt(numel(W))` reaches `||W||_F`. Every sane learning rate
|
|
126
|
+
is a fraction of that, and the fraction is what schedules argue about.
|
|
127
|
+
|
|
128
|
+
Measured, at the ranks behind this library:
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
rewrite/2 2.26e-3 destroyed the model twice (CE 1.74 -> 10.4, 1.11 -> 19.3)
|
|
132
|
+
rewrite/32 1.41e-4 the first scale that actually learned
|
|
133
|
+
rewrite/45 where a hand-tuned LoRA run at 2e-4 landed
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`pichak` opens at `rewrite/32` — the top of the range that works. Opening hot costs
|
|
137
|
+
the model; opening cold costs time. With a penalty that asymmetric you start at the
|
|
138
|
+
bottom of what is known.
|
|
139
|
+
|
|
140
|
+
This is not tuned for your dataset. It is the scale at which updates are the right
|
|
141
|
+
*size* for these weights. It replaces "I copied 2e-4 from a post about a different
|
|
142
|
+
model", which is the actual alternative.
|
|
143
|
+
|
|
144
|
+
## Measurement tools, usable on their own
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
pichak gpu # virtualisation, launch latency, transfer bandwidth
|
|
148
|
+
pichak disk "models/*.safetensors" # queue-depth sweep, page cache bypassed
|
|
149
|
+
pichak corpus train.jsonl mistralai/Mistral-Small-24B-Instruct-2501
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
from pichak.measure import disk_queue_depth, transfer_bandwidth, virtualisation
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
These exist because the numbers people quote are rarely the numbers their machine
|
|
157
|
+
gives:
|
|
158
|
+
|
|
159
|
+
- A buffered disk benchmark reported **4079 MB/s on a drive rated 2100** — that was
|
|
160
|
+
the page cache. Unbuffered, the same drive peaks at **queue depth 3** and gets
|
|
161
|
+
*slower* past it.
|
|
162
|
+
- A rented A6000 measured **D2H 457 MB/s against H2D 1502**, and pinned memory —
|
|
163
|
+
the standard fix — bought nothing. Nothing about PCIe explains that; it was a
|
|
164
|
+
fabric, and it decided where hidden states could live.
|
|
165
|
+
- One rented machine had **84.8us kernel launches** against a normal 2-5. A
|
|
166
|
+
workload issuing 400,000 launches per step would have spent nine hours on
|
|
167
|
+
overhead and looked slower than a 2016 card.
|
|
168
|
+
|
|
169
|
+
## Install
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
pip install pichak # the plan, the disk sweep — no dependencies
|
|
173
|
+
pip install pichak[torch] # the ramp, the learning rate, the GPU measurements
|
|
174
|
+
pip install pichak[hf] # + transformers, for the corpus CLI
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
Python 3.9+. The core has **no dependencies at all**; torch is only needed for the
|
|
178
|
+
parts that touch a GPU.
|
|
179
|
+
|
|
180
|
+
## Where the numbers come from
|
|
181
|
+
|
|
182
|
+
Every measurement quoted here was made while fine-tuning a 24B model on a 6GB
|
|
183
|
+
GTX 1060 and on a rented A6000 held down to the same 6GB. The raw logs — including
|
|
184
|
+
the six runs that failed before one worked — are at
|
|
185
|
+
[flap-findings](https://github.com/olesxg/flap-findings), along with the five
|
|
186
|
+
hypotheses that sounded right and were wrong.
|
|
187
|
+
|
|
188
|
+
## Licence
|
|
189
|
+
|
|
190
|
+
MIT. By [Oleksandr Pichak](https://github.com/olesxg).
|
pichak-0.1.0/README.md
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# pichak
|
|
2
|
+
|
|
3
|
+
**Fine-tuning hyperparameters derived from your machine, with the measurement
|
|
4
|
+
printed next to every number.**
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install pichak
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from pichak import derive
|
|
12
|
+
|
|
13
|
+
plan = derive(
|
|
14
|
+
data="train.jsonl",
|
|
15
|
+
tokenizer=tok,
|
|
16
|
+
step_fn=lambda b: one_real_forward_and_backward(b),
|
|
17
|
+
named_weights=model.named_parameters(),
|
|
18
|
+
)
|
|
19
|
+
print(plan.report())
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
seq_len 512
|
|
24
|
+
token p95 416 over 512 real rows, rounded up to 512; p99 is
|
|
25
|
+
657 and max 943, so 5% of rows lose their tail at this length
|
|
26
|
+
|
|
27
|
+
micro_batch 3
|
|
28
|
+
a doubling ramp with a real forward+backward at each rung:
|
|
29
|
+
1=ok, 2=ok, 4=ok, 8=SPILL. 8 SPILLED to system memory — it did
|
|
30
|
+
not raise, it got 1.6x slower per sample, which is the Windows
|
|
31
|
+
failure mode, so 4 is the largest that fit, peaking at 5.05GB
|
|
32
|
+
of 5.05GB; kept 3 to leave 12% for optimizer state
|
|
33
|
+
|
|
34
|
+
learning_rate 0.000282
|
|
35
|
+
1/32 of the scale at which one Adam step would rewrite rather
|
|
36
|
+
than perturb. ||W||_F/sqrt(numel) over 26 matrices; the
|
|
37
|
+
smallest binds and it is b.1.f.2.weight at 0.00902
|
|
38
|
+
|
|
39
|
+
NOT MEASURED (1): target_tokens=65536
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
## Why not just use a good default
|
|
45
|
+
|
|
46
|
+
Because you cannot tell a good default from a bad one when it fails.
|
|
47
|
+
|
|
48
|
+
`micro_batch=4` tells you nothing about whether 4 came from a measurement on your
|
|
49
|
+
card, a heuristic in a blog post, or a number someone picked in 2023 for a
|
|
50
|
+
different model. So when it OOMs at step 40 you bisect instead of reading.
|
|
51
|
+
|
|
52
|
+
Every number `pichak` returns carries the sentence that produced it. Anything it
|
|
53
|
+
could **not** measure is collected under `plan.constants()` and printed together,
|
|
54
|
+
so a constant can never quietly pass for a derivation.
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
plan.micro_batch # 3
|
|
58
|
+
plan.why("micro_batch") # the ramp, rung by rung, and which one failed
|
|
59
|
+
plan.constants() # {'target_tokens': 65536}
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## It catches the failure that does not raise
|
|
63
|
+
|
|
64
|
+
On Windows, a batch that exceeds VRAM does not throw. The driver pages the excess
|
|
65
|
+
to system memory and the step simply crawls. Measured on a GTX 1060 6GB with a
|
|
66
|
+
6-layer model at sequence 512:
|
|
67
|
+
|
|
68
|
+
| | micro_batch | peak | seconds/step |
|
|
69
|
+
|---|---|---|---|
|
|
70
|
+
| watching only for OOM | 14 | **18.24GB** on a 6GB card | 47.5 |
|
|
71
|
+
| watching per-sample time too | 3 | 5.05GB | **2.0** |
|
|
72
|
+
|
|
73
|
+
Both "work". The first is 23x slower and nothing in the logs says why. On a healthy
|
|
74
|
+
ramp, seconds-per-sample is flat or improves as the batch grows — bigger batches
|
|
75
|
+
amortise fixed costs. A sudden jump means the batch did not really fit.
|
|
76
|
+
|
|
77
|
+
## What it derives, and from what
|
|
78
|
+
|
|
79
|
+
| | measurement |
|
|
80
|
+
|---|---|
|
|
81
|
+
| `seq_len` | tokenised p95 over 512 real rows of your corpus |
|
|
82
|
+
| `loss_policy` | whether the rows have a separable completion to mask against |
|
|
83
|
+
| `micro_batch` | a doubling ramp with a real forward+backward at each rung |
|
|
84
|
+
| `seconds_per_step` | the wall time of that rung |
|
|
85
|
+
| `grad_accum` | micro-batch x seq_len against your token target |
|
|
86
|
+
| `lora_rank` | largest rung whose optimizer state fits — fp32 master, two Adam moments, and one step's gradients, which is the copy count people forget |
|
|
87
|
+
| `learning_rate` | the model's own Frobenius norms: the step size at which one Adam update would rewrite a weight rather than perturb it |
|
|
88
|
+
|
|
89
|
+
Everything is optional. Pass only a corpus and a tokenizer and you get a sequence
|
|
90
|
+
length; pass a `step_fn` and you get the batch. It will not invent a number it
|
|
91
|
+
could not measure — a plan with three derived values and an honest gap is more
|
|
92
|
+
useful than one with ten you cannot tell apart.
|
|
93
|
+
|
|
94
|
+
## The learning rate, since it is the surprising one
|
|
95
|
+
|
|
96
|
+
An Adam update moves a parameter by roughly `lr`, whatever the gradient's scale. So
|
|
97
|
+
there is a learning rate at which one step rewrites a weight matrix instead of
|
|
98
|
+
nudging it: where `lr * sqrt(numel(W))` reaches `||W||_F`. Every sane learning rate
|
|
99
|
+
is a fraction of that, and the fraction is what schedules argue about.
|
|
100
|
+
|
|
101
|
+
Measured, at the ranks behind this library:
|
|
102
|
+
|
|
103
|
+
```
|
|
104
|
+
rewrite/2 2.26e-3 destroyed the model twice (CE 1.74 -> 10.4, 1.11 -> 19.3)
|
|
105
|
+
rewrite/32 1.41e-4 the first scale that actually learned
|
|
106
|
+
rewrite/45 where a hand-tuned LoRA run at 2e-4 landed
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
`pichak` opens at `rewrite/32` — the top of the range that works. Opening hot costs
|
|
110
|
+
the model; opening cold costs time. With a penalty that asymmetric you start at the
|
|
111
|
+
bottom of what is known.
|
|
112
|
+
|
|
113
|
+
This is not tuned for your dataset. It is the scale at which updates are the right
|
|
114
|
+
*size* for these weights. It replaces "I copied 2e-4 from a post about a different
|
|
115
|
+
model", which is the actual alternative.
|
|
116
|
+
|
|
117
|
+
## Measurement tools, usable on their own
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
pichak gpu # virtualisation, launch latency, transfer bandwidth
|
|
121
|
+
pichak disk "models/*.safetensors" # queue-depth sweep, page cache bypassed
|
|
122
|
+
pichak corpus train.jsonl mistralai/Mistral-Small-24B-Instruct-2501
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
from pichak.measure import disk_queue_depth, transfer_bandwidth, virtualisation
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
These exist because the numbers people quote are rarely the numbers their machine
|
|
130
|
+
gives:
|
|
131
|
+
|
|
132
|
+
- A buffered disk benchmark reported **4079 MB/s on a drive rated 2100** — that was
|
|
133
|
+
the page cache. Unbuffered, the same drive peaks at **queue depth 3** and gets
|
|
134
|
+
*slower* past it.
|
|
135
|
+
- A rented A6000 measured **D2H 457 MB/s against H2D 1502**, and pinned memory —
|
|
136
|
+
the standard fix — bought nothing. Nothing about PCIe explains that; it was a
|
|
137
|
+
fabric, and it decided where hidden states could live.
|
|
138
|
+
- One rented machine had **84.8us kernel launches** against a normal 2-5. A
|
|
139
|
+
workload issuing 400,000 launches per step would have spent nine hours on
|
|
140
|
+
overhead and looked slower than a 2016 card.
|
|
141
|
+
|
|
142
|
+
## Install
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
pip install pichak # the plan, the disk sweep — no dependencies
|
|
146
|
+
pip install pichak[torch] # the ramp, the learning rate, the GPU measurements
|
|
147
|
+
pip install pichak[hf] # + transformers, for the corpus CLI
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
Python 3.9+. The core has **no dependencies at all**; torch is only needed for the
|
|
151
|
+
parts that touch a GPU.
|
|
152
|
+
|
|
153
|
+
## Where the numbers come from
|
|
154
|
+
|
|
155
|
+
Every measurement quoted here was made while fine-tuning a 24B model on a 6GB
|
|
156
|
+
GTX 1060 and on a rented A6000 held down to the same 6GB. The raw logs — including
|
|
157
|
+
the six runs that failed before one worked — are at
|
|
158
|
+
[flap-findings](https://github.com/olesxg/flap-findings), along with the five
|
|
159
|
+
hypotheses that sounded right and were wrong.
|
|
160
|
+
|
|
161
|
+
## Licence
|
|
162
|
+
|
|
163
|
+
MIT. By [Oleksandr Pichak](https://github.com/olesxg).
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""pichak — derive fine-tuning hyperparameters from your machine, not from a blog.
|
|
2
|
+
|
|
3
|
+
from pichak import derive
|
|
4
|
+
plan = derive(data="train.jsonl", tokenizer=tok,
|
|
5
|
+
step_fn=lambda b: one_real_step(b),
|
|
6
|
+
named_weights=model.named_parameters())
|
|
7
|
+
print(plan.report())
|
|
8
|
+
|
|
9
|
+
Every number comes back with the measurement that produced it, and anything NOT
|
|
10
|
+
measured is listed together under `plan.constants()` so a constant can never pass
|
|
11
|
+
for a derivation.
|
|
12
|
+
|
|
13
|
+
By Oleksandr Pichak. The measurements this is built on, including the wrong turns,
|
|
14
|
+
are at https://github.com/olesxg/flap-findings
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from .plan import CONSTANT, Plan, Value
|
|
18
|
+
from .tune import (adapter_bytes_per_rank, auto_rank, derive, derive_lr,
|
|
19
|
+
grad_accum_for, profile_corpus, ramp_micro_batch,
|
|
20
|
+
rewrite_scale)
|
|
21
|
+
|
|
22
|
+
__version__ = "0.1.0"
|
|
23
|
+
__author__ = "Oleksandr Pichak"
|
|
24
|
+
__all__ = [
|
|
25
|
+
"derive", "Plan", "Value", "CONSTANT",
|
|
26
|
+
"profile_corpus", "ramp_micro_batch", "auto_rank", "adapter_bytes_per_rank",
|
|
27
|
+
"grad_accum_for", "derive_lr", "rewrite_scale",
|
|
28
|
+
"__version__", "__author__",
|
|
29
|
+
]
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""`pichak` on the command line — measure this machine, print what it gives.
|
|
2
|
+
|
|
3
|
+
pichak gpu virtualisation, launch latency, transfer bandwidth
|
|
4
|
+
pichak disk <glob> queue-depth sweep with the page cache bypassed
|
|
5
|
+
pichak corpus <jsonl> <tokenizer> sequence length from real rows
|
|
6
|
+
"""
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def main(argv=None) -> int:
|
|
11
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
12
|
+
cmd = argv[0] if argv else "help"
|
|
13
|
+
|
|
14
|
+
if cmd == "gpu":
|
|
15
|
+
from .measure import kernel_launch_us, transfer_bandwidth, virtualisation
|
|
16
|
+
v = virtualisation()
|
|
17
|
+
print(f"gpu : {v['gpu']}")
|
|
18
|
+
print(f"virt mode : {v['virtualisation_mode']}")
|
|
19
|
+
print(f"host : {v['host']} ({v['note']})")
|
|
20
|
+
if v["virtualised"]:
|
|
21
|
+
print("\nVIRTUALISED — every timing below is the hypervisor's, not "
|
|
22
|
+
"the card's.")
|
|
23
|
+
try:
|
|
24
|
+
import torch
|
|
25
|
+
except ImportError:
|
|
26
|
+
print("\ntorch is not installed; install pichak[torch] for the rest.")
|
|
27
|
+
return 2
|
|
28
|
+
if not torch.cuda.is_available():
|
|
29
|
+
print("\nno CUDA visible.")
|
|
30
|
+
return 2
|
|
31
|
+
us = kernel_launch_us(torch)
|
|
32
|
+
print(f"\nkernel launch: {us:.1f} us "
|
|
33
|
+
f"(2-5 normal passthrough; 85 seen on a network-attached vGPU)")
|
|
34
|
+
print(" judge this as a SHARE of your step, never in raw microseconds")
|
|
35
|
+
print()
|
|
36
|
+
print(transfer_bandwidth(torch).describe())
|
|
37
|
+
return 0
|
|
38
|
+
|
|
39
|
+
if cmd == "disk":
|
|
40
|
+
if len(argv) < 2:
|
|
41
|
+
print("usage: pichak disk <glob of several large files>")
|
|
42
|
+
return 2
|
|
43
|
+
from .measure import disk_queue_depth
|
|
44
|
+
disk_queue_depth(argv[1:])
|
|
45
|
+
return 0
|
|
46
|
+
|
|
47
|
+
if cmd == "corpus":
|
|
48
|
+
if len(argv) < 3:
|
|
49
|
+
print("usage: pichak corpus <train.jsonl> <tokenizer-name-or-path>")
|
|
50
|
+
return 2
|
|
51
|
+
from transformers import AutoTokenizer
|
|
52
|
+
from .tune import profile_corpus
|
|
53
|
+
tok = AutoTokenizer.from_pretrained(argv[2])
|
|
54
|
+
prof = profile_corpus(argv[1], tok)
|
|
55
|
+
print(prof.describe())
|
|
56
|
+
print(f"\nseq_len {prof.seq_len}\n {prof.seq_source}")
|
|
57
|
+
return 0
|
|
58
|
+
|
|
59
|
+
print(__doc__)
|
|
60
|
+
return 0 if cmd in ("help", "-h", "--help") else 2
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
if __name__ == "__main__":
|
|
64
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Measurements that answer one question each, with no training loop attached.
|
|
2
|
+
|
|
3
|
+
from pichak.measure import disk_queue_depth, transfer_bandwidth
|
|
4
|
+
|
|
5
|
+
These exist because the numbers people quote for storage and PCIe are almost never
|
|
6
|
+
the numbers their machine gives. A buffered disk benchmark on a host with free RAM
|
|
7
|
+
reports the page cache; a rented GPU can have a 3.4x asymmetry between its transfer
|
|
8
|
+
directions that no datasheet mentions.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from .disk import disk_queue_depth
|
|
12
|
+
from .gpu import kernel_launch_us, transfer_bandwidth, virtualisation
|
|
13
|
+
|
|
14
|
+
__all__ = ["disk_queue_depth", "kernel_launch_us", "transfer_bandwidth",
|
|
15
|
+
"virtualisation"]
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""How many concurrent readers does this drive actually want?
|
|
2
|
+
|
|
3
|
+
WHY YOU CANNOT USE A BUFFERED BENCHMARK. The first version of this measurement,
|
|
4
|
+
run with ordinary `open()`, reported:
|
|
5
|
+
|
|
6
|
+
QD1 1158 MB/s QD4 781 MB/s <- slower than one reader
|
|
7
|
+
QD8 4079 MB/s <- above the drive's rated 2100
|
|
8
|
+
|
|
9
|
+
Those are the OS page cache answering, not the disk. If a storage benchmark tells
|
|
10
|
+
you more than the hardware can do, it has stopped measuring the hardware. On
|
|
11
|
+
Windows this bypasses it with `FILE_FLAG_NO_BUFFERING`, which is why the reads have
|
|
12
|
+
to be sector-aligned and why this goes through ctypes.
|
|
13
|
+
|
|
14
|
+
WHAT IT FOUND. On a DRAM-less Kingston NV1:
|
|
15
|
+
|
|
16
|
+
QD1 1017 MB/s QD2 1655 QD3 1703 (optimum) QD4 1535 QD8 1035
|
|
17
|
+
|
|
18
|
+
More threads is not more throughput. A DRAM-less controller degrades past its sweet
|
|
19
|
+
spot rather than saturating, so past QD3 it simply loses — and eight readers, the
|
|
20
|
+
number a reasonable person would pick, is no better than one.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import ctypes
|
|
26
|
+
import glob
|
|
27
|
+
import os
|
|
28
|
+
import sys
|
|
29
|
+
import time
|
|
30
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
31
|
+
from typing import Dict, List, Optional, Sequence
|
|
32
|
+
|
|
33
|
+
MB = 1024 ** 2
|
|
34
|
+
SECTOR = 4096
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _unbuffered_reader():
|
|
38
|
+
"""(reader, available) — an unbuffered read on Windows, else plain reads."""
|
|
39
|
+
if not sys.platform.startswith("win"):
|
|
40
|
+
def posix(path: str, off: int, nbytes: int, block: int) -> int:
|
|
41
|
+
# O_DIRECT needs alignment and is not portable across filesystems;
|
|
42
|
+
# POSIX_FADV_DONTNEED after the read is the honest compromise.
|
|
43
|
+
fd = os.open(path, os.O_RDONLY)
|
|
44
|
+
got = 0
|
|
45
|
+
try:
|
|
46
|
+
os.lseek(fd, off, os.SEEK_SET)
|
|
47
|
+
while got < nbytes:
|
|
48
|
+
b = os.read(fd, min(block, nbytes - got))
|
|
49
|
+
if not b:
|
|
50
|
+
break
|
|
51
|
+
got += len(b)
|
|
52
|
+
if hasattr(os, "posix_fadvise"):
|
|
53
|
+
os.posix_fadvise(fd, off, nbytes, os.POSIX_FADV_DONTNEED)
|
|
54
|
+
finally:
|
|
55
|
+
os.close(fd)
|
|
56
|
+
return got
|
|
57
|
+
return posix, False
|
|
58
|
+
|
|
59
|
+
k32 = ctypes.WinDLL("kernel32", use_last_error=True)
|
|
60
|
+
from ctypes import wintypes
|
|
61
|
+
GENERIC_READ, FILE_SHARE_READ, OPEN_EXISTING = 0x80000000, 1, 3
|
|
62
|
+
NO_BUFFERING, SEQUENTIAL = 0x20000000, 0x08000000
|
|
63
|
+
MEM_COMMIT_RESERVE, PAGE_RW, MEM_RELEASE = 0x3000, 4, 0x8000
|
|
64
|
+
k32.CreateFileW.restype = wintypes.HANDLE
|
|
65
|
+
k32.CreateFileW.argtypes = [wintypes.LPCWSTR, wintypes.DWORD, wintypes.DWORD,
|
|
66
|
+
ctypes.c_void_p, wintypes.DWORD, wintypes.DWORD,
|
|
67
|
+
wintypes.HANDLE]
|
|
68
|
+
k32.VirtualAlloc.restype = ctypes.c_void_p
|
|
69
|
+
k32.VirtualAlloc.argtypes = [ctypes.c_void_p, ctypes.c_size_t,
|
|
70
|
+
wintypes.DWORD, wintypes.DWORD]
|
|
71
|
+
k32.ReadFile.argtypes = [wintypes.HANDLE, ctypes.c_void_p, wintypes.DWORD,
|
|
72
|
+
ctypes.POINTER(wintypes.DWORD), ctypes.c_void_p]
|
|
73
|
+
k32.SetFilePointerEx.argtypes = [wintypes.HANDLE, ctypes.c_longlong,
|
|
74
|
+
ctypes.c_void_p, wintypes.DWORD]
|
|
75
|
+
|
|
76
|
+
def win(path: str, off: int, nbytes: int, block: int) -> int:
|
|
77
|
+
h = k32.CreateFileW(path, GENERIC_READ, FILE_SHARE_READ, None,
|
|
78
|
+
OPEN_EXISTING, NO_BUFFERING | SEQUENTIAL, None)
|
|
79
|
+
if h == wintypes.HANDLE(-1).value:
|
|
80
|
+
raise OSError(ctypes.get_last_error())
|
|
81
|
+
buf = k32.VirtualAlloc(None, block, MEM_COMMIT_RESERVE, PAGE_RW)
|
|
82
|
+
got, done = 0, wintypes.DWORD(0)
|
|
83
|
+
try:
|
|
84
|
+
k32.SetFilePointerEx(h, ctypes.c_longlong(off), None, 0)
|
|
85
|
+
while got < nbytes:
|
|
86
|
+
if not k32.ReadFile(h, ctypes.c_void_p(buf), block,
|
|
87
|
+
ctypes.byref(done), None):
|
|
88
|
+
raise OSError(ctypes.get_last_error())
|
|
89
|
+
if done.value == 0:
|
|
90
|
+
break
|
|
91
|
+
got += done.value
|
|
92
|
+
finally:
|
|
93
|
+
k32.VirtualFree(ctypes.c_void_p(buf), 0, MEM_RELEASE)
|
|
94
|
+
k32.CloseHandle(h)
|
|
95
|
+
return got
|
|
96
|
+
return win, True
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def disk_queue_depth(paths: Sequence[str], *, depths: Sequence[int] = (1, 2, 3, 4, 6, 8),
|
|
100
|
+
chunk_mb: int = 256, block_mb: int = 8,
|
|
101
|
+
log=print) -> Dict[int, float]:
|
|
102
|
+
"""MB/s at each queue depth. Returns {depth: rate}, and prints as it goes.
|
|
103
|
+
|
|
104
|
+
``paths`` should be several large files — model shards are ideal. Workers read
|
|
105
|
+
DIFFERENT files at DIFFERENT offsets so that a second reader cannot be served
|
|
106
|
+
from the first one's readahead.
|
|
107
|
+
"""
|
|
108
|
+
files = [p for pat in paths for p in sorted(glob.glob(pat))] if any(
|
|
109
|
+
"*" in p or "?" in p for p in paths) else list(paths)
|
|
110
|
+
files = [f for f in files if os.path.isfile(f)]
|
|
111
|
+
if not files:
|
|
112
|
+
raise FileNotFoundError(f"no readable files in {list(paths)!r}")
|
|
113
|
+
|
|
114
|
+
read, unbuffered = _unbuffered_reader()
|
|
115
|
+
if not unbuffered:
|
|
116
|
+
log(" note: unbuffered reads are Windows-only here; on this platform the "
|
|
117
|
+
"page cache is advised away after each read, which is weaker. Treat "
|
|
118
|
+
"any result above the drive's rating as cache.")
|
|
119
|
+
|
|
120
|
+
chunk, block = chunk_mb * MB, block_mb * MB
|
|
121
|
+
out: Dict[int, float] = {}
|
|
122
|
+
log(f" {len(files)} file(s), {chunk_mb}MB per worker, {block_mb}MB blocks")
|
|
123
|
+
for d in depths:
|
|
124
|
+
jobs = []
|
|
125
|
+
for i in range(d):
|
|
126
|
+
p = files[i % len(files)]
|
|
127
|
+
size = (os.path.getsize(p) // SECTOR) * SECTOR
|
|
128
|
+
slot = i // len(files)
|
|
129
|
+
off = ((slot * 1300 * MB + 300 * MB) % max(SECTOR, size - chunk))
|
|
130
|
+
jobs.append((p, (off // SECTOR) * SECTOR, chunk, block))
|
|
131
|
+
t0 = time.perf_counter()
|
|
132
|
+
with ThreadPoolExecutor(max_workers=d) as ex:
|
|
133
|
+
total = sum(ex.map(lambda a: read(*a), jobs))
|
|
134
|
+
rate = total / (time.perf_counter() - t0) / MB
|
|
135
|
+
out[d] = rate
|
|
136
|
+
log(f" QD{d:<3d} {total / MB:6.0f} MB in {(total / MB) / rate:5.2f}s "
|
|
137
|
+
f"= {rate:6.0f} MB/s")
|
|
138
|
+
best = max(out, key=out.get)
|
|
139
|
+
log(f" optimum QD{best} at {out[best]:.0f} MB/s "
|
|
140
|
+
f"({out[best] / out[min(out)]:.2f}x QD{min(out)})")
|
|
141
|
+
if best < max(out):
|
|
142
|
+
log(" more readers than that LOSE — a DRAM-less controller degrades past "
|
|
143
|
+
"its sweet spot rather than saturating.")
|
|
144
|
+
return out
|