faster-diffbloch 0.1.2__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/PKG-INFO +48 -28
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/README.md +46 -26
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/pyproject.toml +2 -2
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/__init__.py +1 -1
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/backend.py +37 -2
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/.gitignore +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/LICENSE +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/builder.py +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/cli.py +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/batch_cgemm.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/batch_cgemm.h +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/batch_cgemm.metal +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/bridge_lib.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/metal_batch_cgemm.m +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/native_scattering.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/native_scattering.h +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: faster-diffbloch
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Summary: Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch
|
|
5
5
|
Project-URL: Homepage, https://godofecht.github.io/diffFlow/
|
|
6
6
|
Project-URL: Documentation, https://godofecht.github.io/diffFlow/
|
|
@@ -9,7 +9,7 @@ Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
|
|
|
9
9
|
Project-URL: PyPI, https://pypi.org/project/faster-diffbloch/
|
|
10
10
|
Project-URL: piwheels, https://www.piwheels.org/project/faster-diffbloch/
|
|
11
11
|
Project-URL: Original diffBloch, https://diffbloch.com
|
|
12
|
-
Author-email: Abhishek Shivakumar <abhishek@
|
|
12
|
+
Author-email: Abhishek Shivakumar <abhishek.shivakumar@gmail.com>
|
|
13
13
|
License-Expression: MIT
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Keywords: Metal GPU,PyTorch acceleration,diffBloch,electron crystallography,electron diffraction,structure refinement
|
|
@@ -35,7 +35,7 @@ Description-Content-Type: text/markdown
|
|
|
35
35
|
|
|
36
36
|
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
37
37
|
|
|
38
|
-
|
|
38
|
+
A Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux, measured against diffBloch running on PyTorch.
|
|
39
39
|
|
|
40
40
|
| Package | Documentation | Source repository | Original project |
|
|
41
41
|
| :--- | :--- | :--- | :--- |
|
|
@@ -63,38 +63,58 @@ the expensive propagation and gradient work onto an accelerated native backend.
|
|
|
63
63
|
You get the same scientific calculation, with a faster path on supported hardware
|
|
64
64
|
and a safe PyTorch fallback when the native backend is unavailable.
|
|
65
65
|
|
|
66
|
-
The package is validated against
|
|
67
|
-
|
|
68
|
-
the
|
|
66
|
+
The package is validated against diffBloch's own test suite. With acceleration
|
|
67
|
+
enabled it passes all 738 diffBloch tests and 55 conformance tests that compare
|
|
68
|
+
the accelerated results against diffBloch's, field by field, including gradients.
|
|
69
|
+
The diffBloch run makes 4855 calls into the native matrix exponential, so the
|
|
70
|
+
accelerated path is exercised rather than skipped.
|
|
69
71
|
|
|
70
72
|
---
|
|
71
73
|
|
|
72
74
|
## Performance
|
|
73
75
|
|
|
74
|
-
|
|
75
|
-
same diffBloch workload
|
|
76
|
+
Forward-plus-backward timing on an Apple M4 Max, single-threaded PyTorch,
|
|
77
|
+
running the same diffBloch workload through `bench/compare_faster.py`. This is
|
|
78
|
+
what `pip install faster-diffbloch` gives you: the accelerated matrix
|
|
79
|
+
exponential bridged into diffBloch, with PyTorch handling everything else.
|
|
76
80
|
|
|
77
|
-
| Beams | faster-
|
|
81
|
+
| Beams | faster-diffbloch | PyTorch | Speedup |
|
|
78
82
|
| :---: | :---: | :---: | :---: |
|
|
79
|
-
| 31 |
|
|
80
|
-
| 61 | 1.
|
|
81
|
-
| 91 |
|
|
82
|
-
| 163 |
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
83
|
+
| 31 | 1.48 ms | 1.54 ms | 1.05x |
|
|
84
|
+
| 61 | 1.93 ms | 2.07 ms | 1.07x |
|
|
85
|
+
| 91 | 3.48 ms | 3.69 ms | 1.06x |
|
|
86
|
+
| 163 | 8.76 ms | 10.14 ms | 1.16x |
|
|
87
|
+
|
|
88
|
+
Best of three alternating runs per mode. The gain grows with beam count,
|
|
89
|
+
because the matrix exponential takes a larger share of the work as the system
|
|
90
|
+
grows and the fixed ctypes and PyTorch overhead matters less.
|
|
91
|
+
|
|
92
|
+
### The standalone Flow port
|
|
93
|
+
|
|
94
|
+
The diffFlow repository also holds a standalone port of the whole calculation,
|
|
95
|
+
compiled from Flow rather than bridged into PyTorch. It avoids the framework
|
|
96
|
+
overhead entirely and is considerably faster. At 579 beams, forward plus
|
|
97
|
+
backward, minimum of five runs:
|
|
98
|
+
|
|
99
|
+
| Implementation | Forward + Backward |
|
|
100
|
+
| :--- | :---: |
|
|
101
|
+
| PyTorch | 117.21 ms |
|
|
102
|
+
| Mojo port | 85.52 ms |
|
|
103
|
+
| Flow port, C backend | 82.84 ms |
|
|
104
|
+
| Flow port, MLIR backend | 79.74 ms |
|
|
105
|
+
|
|
106
|
+
Those numbers are not what this package delivers. They are reachable by
|
|
107
|
+
running the port directly, and they are why the package exists. See the
|
|
108
|
+
[diffFlow documentation](https://godofecht.github.io/diffFlow/).
|
|
109
|
+
|
|
110
|
+
These figures come from one workload on one machine. Other crystals, hardware
|
|
111
|
+
configurations and beam counts will differ.
|
|
112
|
+
|
|
113
|
+
### Metal GPU
|
|
114
|
+
|
|
115
|
+
The Metal path is available on Apple Silicon through `enable(device="gpu")`.
|
|
116
|
+
The table below is from an earlier run and has not been reproduced against the
|
|
117
|
+
current package, so treat it as indicative.
|
|
98
118
|
|
|
99
119
|
### Package benchmark snapshot
|
|
100
120
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
A Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux, measured against diffBloch running on PyTorch.
|
|
6
6
|
|
|
7
7
|
| Package | Documentation | Source repository | Original project |
|
|
8
8
|
| :--- | :--- | :--- | :--- |
|
|
@@ -30,38 +30,58 @@ the expensive propagation and gradient work onto an accelerated native backend.
|
|
|
30
30
|
You get the same scientific calculation, with a faster path on supported hardware
|
|
31
31
|
and a safe PyTorch fallback when the native backend is unavailable.
|
|
32
32
|
|
|
33
|
-
The package is validated against
|
|
34
|
-
|
|
35
|
-
the
|
|
33
|
+
The package is validated against diffBloch's own test suite. With acceleration
|
|
34
|
+
enabled it passes all 738 diffBloch tests and 55 conformance tests that compare
|
|
35
|
+
the accelerated results against diffBloch's, field by field, including gradients.
|
|
36
|
+
The diffBloch run makes 4855 calls into the native matrix exponential, so the
|
|
37
|
+
accelerated path is exercised rather than skipped.
|
|
36
38
|
|
|
37
39
|
---
|
|
38
40
|
|
|
39
41
|
## Performance
|
|
40
42
|
|
|
41
|
-
|
|
42
|
-
same diffBloch workload
|
|
43
|
+
Forward-plus-backward timing on an Apple M4 Max, single-threaded PyTorch,
|
|
44
|
+
running the same diffBloch workload through `bench/compare_faster.py`. This is
|
|
45
|
+
what `pip install faster-diffbloch` gives you: the accelerated matrix
|
|
46
|
+
exponential bridged into diffBloch, with PyTorch handling everything else.
|
|
43
47
|
|
|
44
|
-
| Beams | faster-
|
|
48
|
+
| Beams | faster-diffbloch | PyTorch | Speedup |
|
|
45
49
|
| :---: | :---: | :---: | :---: |
|
|
46
|
-
| 31 |
|
|
47
|
-
| 61 | 1.
|
|
48
|
-
| 91 |
|
|
49
|
-
| 163 |
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
50
|
+
| 31 | 1.48 ms | 1.54 ms | 1.05x |
|
|
51
|
+
| 61 | 1.93 ms | 2.07 ms | 1.07x |
|
|
52
|
+
| 91 | 3.48 ms | 3.69 ms | 1.06x |
|
|
53
|
+
| 163 | 8.76 ms | 10.14 ms | 1.16x |
|
|
54
|
+
|
|
55
|
+
Best of three alternating runs per mode. The gain grows with beam count,
|
|
56
|
+
because the matrix exponential takes a larger share of the work as the system
|
|
57
|
+
grows and the fixed ctypes and PyTorch overhead matters less.
|
|
58
|
+
|
|
59
|
+
### The standalone Flow port
|
|
60
|
+
|
|
61
|
+
The diffFlow repository also holds a standalone port of the whole calculation,
|
|
62
|
+
compiled from Flow rather than bridged into PyTorch. It avoids the framework
|
|
63
|
+
overhead entirely and is considerably faster. At 579 beams, forward plus
|
|
64
|
+
backward, minimum of five runs:
|
|
65
|
+
|
|
66
|
+
| Implementation | Forward + Backward |
|
|
67
|
+
| :--- | :---: |
|
|
68
|
+
| PyTorch | 117.21 ms |
|
|
69
|
+
| Mojo port | 85.52 ms |
|
|
70
|
+
| Flow port, C backend | 82.84 ms |
|
|
71
|
+
| Flow port, MLIR backend | 79.74 ms |
|
|
72
|
+
|
|
73
|
+
Those numbers are not what this package delivers. They are reachable by
|
|
74
|
+
running the port directly, and they are why the package exists. See the
|
|
75
|
+
[diffFlow documentation](https://godofecht.github.io/diffFlow/).
|
|
76
|
+
|
|
77
|
+
These figures come from one workload on one machine. Other crystals, hardware
|
|
78
|
+
configurations and beam counts will differ.
|
|
79
|
+
|
|
80
|
+
### Metal GPU
|
|
81
|
+
|
|
82
|
+
The Metal path is available on Apple Silicon through `enable(device="gpu")`.
|
|
83
|
+
The table below is from an earlier run and has not been reproduced against the
|
|
84
|
+
current package, so treat it as indicative.
|
|
65
85
|
|
|
66
86
|
### Package benchmark snapshot
|
|
67
87
|
|
|
@@ -4,14 +4,14 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "faster-diffbloch"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.4"
|
|
8
8
|
description = "Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
keywords = ["diffBloch", "electron diffraction", "electron crystallography", "structure refinement", "PyTorch acceleration", "Metal GPU"]
|
|
12
12
|
license = "MIT"
|
|
13
13
|
authors = [
|
|
14
|
-
{ name = "Abhishek Shivakumar", email = "abhishek@
|
|
14
|
+
{ name = "Abhishek Shivakumar", email = "abhishek.shivakumar@gmail.com" }
|
|
15
15
|
]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 4 - Beta",
|
|
@@ -20,6 +20,11 @@ from .builder import (
|
|
|
20
20
|
_LIB: ctypes.CDLL | None = None
|
|
21
21
|
_CURRENT_DEVICE: Literal["cpu", "gpu"] = "gpu"
|
|
22
22
|
|
|
23
|
+
# What enable() replaced, keyed by (module name, attribute), so disable() can
|
|
24
|
+
# put it back. A sentinel marks an attribute that did not exist before.
|
|
25
|
+
_MISSING = object()
|
|
26
|
+
_ORIGINALS: dict[tuple[str, str], object] = {}
|
|
27
|
+
|
|
23
28
|
|
|
24
29
|
def get_library() -> ctypes.CDLL | None:
|
|
25
30
|
global _LIB
|
|
@@ -181,11 +186,14 @@ def enable(device: Literal["cpu", "gpu"] = "gpu") -> None:
|
|
|
181
186
|
import importlib
|
|
182
187
|
try:
|
|
183
188
|
solver = importlib.import_module("diffBloch.core.solver")
|
|
189
|
+
_remember(solver, "_propagate_matrix_exp")
|
|
190
|
+
_remember(solver, "propagate")
|
|
184
191
|
solver._propagate_matrix_exp = faster_propagate_matrix_exp
|
|
185
192
|
solver.propagate = faster_propagate
|
|
186
193
|
for mod_name in ("diffBloch.core", "diffBloch.engine.forward"):
|
|
187
194
|
try:
|
|
188
195
|
mod = importlib.import_module(mod_name)
|
|
196
|
+
_remember(mod, "propagate")
|
|
189
197
|
setattr(mod, "propagate", faster_propagate)
|
|
190
198
|
except Exception:
|
|
191
199
|
pass
|
|
@@ -193,6 +201,33 @@ def enable(device: Literal["cpu", "gpu"] = "gpu") -> None:
|
|
|
193
201
|
pass
|
|
194
202
|
|
|
195
203
|
|
|
204
|
+
def _remember(module, name: str) -> None:
|
|
205
|
+
"""Record a module attribute the first time it is replaced.
|
|
206
|
+
|
|
207
|
+
Only the first value is kept, so calling enable() twice does not record
|
|
208
|
+
the accelerated function as the thing to restore.
|
|
209
|
+
"""
|
|
210
|
+
key = (module.__name__, name)
|
|
211
|
+
if key not in _ORIGINALS:
|
|
212
|
+
_ORIGINALS[key] = getattr(module, name, _MISSING)
|
|
213
|
+
|
|
214
|
+
|
|
196
215
|
def disable() -> None:
|
|
197
|
-
"""
|
|
198
|
-
|
|
216
|
+
"""Put diffBloch's own functions back.
|
|
217
|
+
|
|
218
|
+
Undoes what enable() replaced, so a later call runs stock diffBloch.
|
|
219
|
+
Calling this without a preceding enable() does nothing.
|
|
220
|
+
"""
|
|
221
|
+
import importlib
|
|
222
|
+
|
|
223
|
+
for (mod_name, attr), original in list(_ORIGINALS.items()):
|
|
224
|
+
try:
|
|
225
|
+
module = importlib.import_module(mod_name)
|
|
226
|
+
except ImportError:
|
|
227
|
+
continue
|
|
228
|
+
if original is _MISSING:
|
|
229
|
+
if hasattr(module, attr):
|
|
230
|
+
delattr(module, attr)
|
|
231
|
+
else:
|
|
232
|
+
setattr(module, attr, original)
|
|
233
|
+
_ORIGINALS.clear()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/batch_cgemm.metal
RENAMED
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/metal_batch_cgemm.m
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/native_scattering.c
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.4}/src/faster_diffbloch/native/native_scattering.h
RENAMED
|
File without changes
|