faster-diffbloch 0.1.2__tar.gz → 0.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/PKG-INFO +5 -5
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/README.md +3 -3
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/pyproject.toml +2 -2
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/__init__.py +1 -1
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/.gitignore +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/LICENSE +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/backend.py +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/builder.py +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/cli.py +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/batch_cgemm.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/batch_cgemm.h +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/batch_cgemm.metal +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/bridge_lib.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/metal_batch_cgemm.m +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/native_scattering.c +0 -0
- {faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/native_scattering.h +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: faster-diffbloch
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.3
|
|
4
4
|
Summary: Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch
|
|
5
5
|
Project-URL: Homepage, https://godofecht.github.io/diffFlow/
|
|
6
6
|
Project-URL: Documentation, https://godofecht.github.io/diffFlow/
|
|
@@ -9,7 +9,7 @@ Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
|
|
|
9
9
|
Project-URL: PyPI, https://pypi.org/project/faster-diffbloch/
|
|
10
10
|
Project-URL: piwheels, https://www.piwheels.org/project/faster-diffbloch/
|
|
11
11
|
Project-URL: Original diffBloch, https://diffbloch.com
|
|
12
|
-
Author-email: Abhishek Shivakumar <abhishek@
|
|
12
|
+
Author-email: Abhishek Shivakumar <abhishek.shivakumar@gmail.com>
|
|
13
13
|
License-Expression: MIT
|
|
14
14
|
License-File: LICENSE
|
|
15
15
|
Keywords: Metal GPU,PyTorch acceleration,diffBloch,electron crystallography,electron diffraction,structure refinement
|
|
@@ -82,7 +82,7 @@ same diffBloch workload at different beam counts:
|
|
|
82
82
|
| 163 | 7.67 ms | 10.98 ms | **1.43×** |
|
|
83
83
|
| 579 | 89.81 ms | 135.77 ms | **1.51×** |
|
|
84
84
|
|
|
85
|
-
That is **1.
|
|
85
|
+
That is **1.41x to 1.80x faster than PyTorch CPU** across the measured cases,
|
|
86
86
|
including the largest 579-beam case. The same benchmark is **1.21× faster than
|
|
87
87
|
the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
|
|
88
88
|
essentially level at 579 beams.
|
|
@@ -91,8 +91,8 @@ On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
|
|
|
91
91
|
comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
|
|
92
92
|
forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
|
|
93
93
|
|
|
94
|
-
These figures
|
|
95
|
-
hardware
|
|
94
|
+
These figures come from one workload on one machine. Other crystals,
|
|
95
|
+
hardware configurations and beam counts will differ. The CPU table
|
|
96
96
|
uses single-threaded PyTorch and the same Apple Accelerate environment for a
|
|
97
97
|
like-for-like comparison.
|
|
98
98
|
|
|
@@ -49,7 +49,7 @@ same diffBloch workload at different beam counts:
|
|
|
49
49
|
| 163 | 7.67 ms | 10.98 ms | **1.43×** |
|
|
50
50
|
| 579 | 89.81 ms | 135.77 ms | **1.51×** |
|
|
51
51
|
|
|
52
|
-
That is **1.
|
|
52
|
+
That is **1.41x to 1.80x faster than PyTorch CPU** across the measured cases,
|
|
53
53
|
including the largest 579-beam case. The same benchmark is **1.21× faster than
|
|
54
54
|
the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
|
|
55
55
|
essentially level at 579 beams.
|
|
@@ -58,8 +58,8 @@ On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
|
|
|
58
58
|
comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
|
|
59
59
|
forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
|
|
60
60
|
|
|
61
|
-
These figures
|
|
62
|
-
hardware
|
|
61
|
+
These figures come from one workload on one machine. Other crystals,
|
|
62
|
+
hardware configurations and beam counts will differ. The CPU table
|
|
63
63
|
uses single-threaded PyTorch and the same Apple Accelerate environment for a
|
|
64
64
|
like-for-like comparison.
|
|
65
65
|
|
|
@@ -4,14 +4,14 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "faster-diffbloch"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.3"
|
|
8
8
|
description = "Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
keywords = ["diffBloch", "electron diffraction", "electron crystallography", "structure refinement", "PyTorch acceleration", "Metal GPU"]
|
|
12
12
|
license = "MIT"
|
|
13
13
|
authors = [
|
|
14
|
-
{ name = "Abhishek Shivakumar", email = "abhishek@
|
|
14
|
+
{ name = "Abhishek Shivakumar", email = "abhishek.shivakumar@gmail.com" }
|
|
15
15
|
]
|
|
16
16
|
classifiers = [
|
|
17
17
|
"Development Status :: 4 - Beta",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/batch_cgemm.metal
RENAMED
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/metal_batch_cgemm.m
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/native_scattering.c
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.2 → faster_diffbloch-0.1.3}/src/faster_diffbloch/native/native_scattering.h
RENAMED
|
File without changes
|