faster-diffbloch 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (19) hide show
  1. faster_diffbloch-0.1.2/PKG-INFO +145 -0
  2. faster_diffbloch-0.1.2/README.md +112 -0
  3. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/pyproject.toml +9 -2
  4. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/__init__.py +1 -1
  5. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/backend.py +70 -21
  6. faster_diffbloch-0.1.2/src/faster_diffbloch/builder.py +122 -0
  7. faster_diffbloch-0.1.0/PKG-INFO +0 -98
  8. faster_diffbloch-0.1.0/README.md +0 -72
  9. faster_diffbloch-0.1.0/src/faster_diffbloch/builder.py +0 -58
  10. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/.gitignore +0 -0
  11. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/LICENSE +0 -0
  12. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/cli.py +0 -0
  13. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.c +0 -0
  14. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.h +0 -0
  15. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.metal +0 -0
  16. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/bridge_lib.c +0 -0
  17. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/metal_batch_cgemm.m +0 -0
  18. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.c +0 -0
  19. {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.h +0 -0
@@ -0,0 +1,145 @@
1
+ Metadata-Version: 2.5
2
+ Name: faster-diffbloch
3
+ Version: 0.1.2
4
+ Summary: Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch
5
+ Project-URL: Homepage, https://godofecht.github.io/diffFlow/
6
+ Project-URL: Documentation, https://godofecht.github.io/diffFlow/
7
+ Project-URL: Repository, https://github.com/godofecht/diffFlow
8
+ Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
9
+ Project-URL: PyPI, https://pypi.org/project/faster-diffbloch/
10
+ Project-URL: piwheels, https://www.piwheels.org/project/faster-diffbloch/
11
+ Project-URL: Original diffBloch, https://diffbloch.com
12
+ Author-email: Abhishek Shivakumar <abhishek@example.com>
13
+ License-Expression: MIT
14
+ License-File: LICENSE
15
+ Keywords: Metal GPU,PyTorch acceleration,diffBloch,electron crystallography,electron diffraction,structure refinement
16
+ Classifier: Development Status :: 4 - Beta
17
+ Classifier: Intended Audience :: Science/Research
18
+ Classifier: License :: OSI Approved :: MIT License
19
+ Classifier: Operating System :: MacOS
20
+ Classifier: Operating System :: MacOS :: MacOS X
21
+ Classifier: Operating System :: POSIX :: Linux
22
+ Classifier: Programming Language :: Python :: 3
23
+ Classifier: Programming Language :: Python :: 3.10
24
+ Classifier: Programming Language :: Python :: 3.11
25
+ Classifier: Programming Language :: Python :: 3.12
26
+ Classifier: Topic :: Scientific/Engineering :: Physics
27
+ Requires-Python: >=3.10
28
+ Requires-Dist: numpy>=1.24
29
+ Requires-Dist: torch>=2.0
30
+ Provides-Extra: diffbloch
31
+ Requires-Dist: diffbloch; extra == 'diffbloch'
32
+ Description-Content-Type: text/markdown
33
+
34
+ # faster-diffbloch
35
+
36
+ Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
37
+
38
+ **Up to 1.80× faster than PyTorch CPU** in measured end-to-end forward-plus-backward benchmarks, with a Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux.
39
+
40
+ | Package | Documentation | Source repository | Original project |
41
+ | :--- | :--- | :--- | :--- |
42
+ | [PyPI](https://pypi.org/project/faster-diffbloch/) · [piwheels](https://www.piwheels.org/project/faster-diffbloch/) | [diffFlow documentation and benchmarks](https://godofecht.github.io/diffFlow/) | [github.com/godofecht/diffFlow](https://github.com/godofecht/diffFlow) | [diffbloch.com](https://diffbloch.com) |
43
+
44
+ ---
45
+
46
+ ## Operating System and Platform Support
47
+
48
+ | Operating System / Hardware | CPU Acceleration (`device="cpu"`) | GPU Acceleration (`device="gpu"`) | Backend Runtime |
49
+ | :--- | :---: | :---: | :--- |
50
+ | **macOS Apple Silicon (M1/M2/M3/M4/Max/Ultra)** | **Supported** | **Supported** | Native Metal Compute Shaders + Apple Accelerate BLAS |
51
+ | **macOS Intel (x86_64)** | **Supported** | Fallback to CPU | Apple Accelerate BLAS |
52
+ | **Linux (x86_64 / aarch64)** | **Supported** | Fallback to CPU | OpenBLAS / C11 BLAS |
53
+ | **Windows** | PyTorch Reference | PyTorch Reference | Pure PyTorch Reference Fallback |
54
+
55
+ Runtime platform guards automatically detect your operating system and hardware configuration. When `device="gpu"` is requested on Linux, `faster-diffbloch` automatically selects the optimized CPU backend with an informative warning.
56
+
57
+ ---
58
+
59
+ ## Why faster-diffBloch?
60
+
61
+ `faster-diffbloch` keeps the diffBloch workflow and public API intact while moving
62
+ the expensive propagation and gradient work onto an accelerated native backend.
63
+ You get the same scientific calculation, with a faster path on supported hardware
64
+ and a safe PyTorch fallback when the native backend is unavailable.
65
+
66
+ The package is validated against the diffBloch test suite and reference results.
67
+ It passes all 738 diffBloch tests, 55 additional conformance tests, and reproduces
68
+ the experimental quartz refinement result.
69
+
70
+ ---
71
+
72
+ ## Performance
73
+
74
+ End-to-end forward-plus-backward timing on an Apple M4 Max, measured across the
75
+ same diffBloch workload at different beam counts:
76
+
77
+ | Beams | faster-diffBloch | PyTorch | Speedup |
78
+ | :---: | :---: | :---: | :---: |
79
+ | 31 | 0.90 ms | 1.62 ms | **1.80×** |
80
+ | 61 | 1.37 ms | 2.24 ms | **1.63×** |
81
+ | 91 | 2.84 ms | 3.99 ms | **1.41×** |
82
+ | 163 | 7.67 ms | 10.98 ms | **1.43×** |
83
+ | 579 | 89.81 ms | 135.77 ms | **1.51×** |
84
+
85
+ That is **1.41×–1.80× faster than PyTorch CPU** across the measured cases,
86
+ including the largest 579-beam case. The same benchmark is **1.21× faster than
87
+ the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
88
+ essentially level at 579 beams.
89
+
90
+ On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
91
+ comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
92
+ forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
93
+
94
+ These figures are workload benchmarks, not a promise that every crystal,
95
+ hardware configuration, or beam count will see the same result. The CPU table
96
+ uses single-threaded PyTorch and the same Apple Accelerate environment for a
97
+ like-for-like comparison.
98
+
99
+ ### Package benchmark snapshot
100
+
101
+ | Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
102
+ | :--- | :---: | :---: | :---: | :---: |
103
+ | PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
104
+ | PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
105
+ | **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
106
+ | **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
107
+
108
+ ---
109
+
110
+ ## Installation
111
+
112
+ ```bash
113
+ pip install faster-diffbloch
114
+ ```
115
+
116
+ ---
117
+
118
+ ## Usage
119
+
120
+ ### 1. Drop-in CLI
121
+
122
+ Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
123
+
124
+ ```bash
125
+ diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
126
+ diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
127
+ ```
128
+
129
+ ### 2. Python API Injection
130
+
131
+ Enable acceleration inside any existing diffBloch script:
132
+
133
+ ```python
134
+ import faster_diffbloch
135
+
136
+ # Enable Metal GPU acceleration (macOS Apple Silicon)
137
+ faster_diffbloch.enable(device="gpu")
138
+
139
+ # Or CPU acceleration (macOS and Linux)
140
+ faster_diffbloch.enable(device="cpu")
141
+
142
+ # Run standard diffBloch code
143
+ import diffBloch
144
+ # All propagate and matrix_exp calls now route through faster-diffBloch
145
+ ```
@@ -0,0 +1,112 @@
1
+ # faster-diffbloch
2
+
3
+ Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
4
+
5
+ **Up to 1.80× faster than PyTorch CPU** in measured end-to-end forward-plus-backward benchmarks, with a Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux.
6
+
7
+ | Package | Documentation | Source repository | Original project |
8
+ | :--- | :--- | :--- | :--- |
9
+ | [PyPI](https://pypi.org/project/faster-diffbloch/) · [piwheels](https://www.piwheels.org/project/faster-diffbloch/) | [diffFlow documentation and benchmarks](https://godofecht.github.io/diffFlow/) | [github.com/godofecht/diffFlow](https://github.com/godofecht/diffFlow) | [diffbloch.com](https://diffbloch.com) |
10
+
11
+ ---
12
+
13
+ ## Operating System and Platform Support
14
+
15
+ | Operating System / Hardware | CPU Acceleration (`device="cpu"`) | GPU Acceleration (`device="gpu"`) | Backend Runtime |
16
+ | :--- | :---: | :---: | :--- |
17
+ | **macOS Apple Silicon (M1/M2/M3/M4/Max/Ultra)** | **Supported** | **Supported** | Native Metal Compute Shaders + Apple Accelerate BLAS |
18
+ | **macOS Intel (x86_64)** | **Supported** | Fallback to CPU | Apple Accelerate BLAS |
19
+ | **Linux (x86_64 / aarch64)** | **Supported** | Fallback to CPU | OpenBLAS / C11 BLAS |
20
+ | **Windows** | PyTorch Reference | PyTorch Reference | Pure PyTorch Reference Fallback |
21
+
22
+ Runtime platform guards automatically detect your operating system and hardware configuration. When `device="gpu"` is requested on Linux, `faster-diffbloch` automatically selects the optimized CPU backend with an informative warning.
23
+
24
+ ---
25
+
26
+ ## Why faster-diffBloch?
27
+
28
+ `faster-diffbloch` keeps the diffBloch workflow and public API intact while moving
29
+ the expensive propagation and gradient work onto an accelerated native backend.
30
+ You get the same scientific calculation, with a faster path on supported hardware
31
+ and a safe PyTorch fallback when the native backend is unavailable.
32
+
33
+ The package is validated against the diffBloch test suite and reference results.
34
+ It passes all 738 diffBloch tests, 55 additional conformance tests, and reproduces
35
+ the experimental quartz refinement result.
36
+
37
+ ---
38
+
39
+ ## Performance
40
+
41
+ End-to-end forward-plus-backward timing on an Apple M4 Max, measured across the
42
+ same diffBloch workload at different beam counts:
43
+
44
+ | Beams | faster-diffBloch | PyTorch | Speedup |
45
+ | :---: | :---: | :---: | :---: |
46
+ | 31 | 0.90 ms | 1.62 ms | **1.80×** |
47
+ | 61 | 1.37 ms | 2.24 ms | **1.63×** |
48
+ | 91 | 2.84 ms | 3.99 ms | **1.41×** |
49
+ | 163 | 7.67 ms | 10.98 ms | **1.43×** |
50
+ | 579 | 89.81 ms | 135.77 ms | **1.51×** |
51
+
52
+ That is **1.41×–1.80× faster than PyTorch CPU** across the measured cases,
53
+ including the largest 579-beam case. The same benchmark is **1.21× faster than
54
+ the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
55
+ essentially level at 579 beams.
56
+
57
+ On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
58
+ comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
59
+ forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
60
+
61
+ These figures are workload benchmarks, not a promise that every crystal,
62
+ hardware configuration, or beam count will see the same result. The CPU table
63
+ uses single-threaded PyTorch and the same Apple Accelerate environment for a
64
+ like-for-like comparison.
65
+
66
+ ### Package benchmark snapshot
67
+
68
+ | Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
69
+ | :--- | :---: | :---: | :---: | :---: |
70
+ | PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
71
+ | PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
72
+ | **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
73
+ | **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
74
+
75
+ ---
76
+
77
+ ## Installation
78
+
79
+ ```bash
80
+ pip install faster-diffbloch
81
+ ```
82
+
83
+ ---
84
+
85
+ ## Usage
86
+
87
+ ### 1. Drop-in CLI
88
+
89
+ Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
90
+
91
+ ```bash
92
+ diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
93
+ diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
94
+ ```
95
+
96
+ ### 2. Python API Injection
97
+
98
+ Enable acceleration inside any existing diffBloch script:
99
+
100
+ ```python
101
+ import faster_diffbloch
102
+
103
+ # Enable Metal GPU acceleration (macOS Apple Silicon)
104
+ faster_diffbloch.enable(device="gpu")
105
+
106
+ # Or CPU acceleration (macOS and Linux)
107
+ faster_diffbloch.enable(device="cpu")
108
+
109
+ # Run standard diffBloch code
110
+ import diffBloch
111
+ # All propagate and matrix_exp calls now route through faster-diffBloch
112
+ ```
@@ -4,10 +4,11 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "faster-diffbloch"
7
- version = "0.1.0"
8
- description = "Drop-in Metal GPU and CPU acceleration for diffBloch electron crystallography"
7
+ version = "0.1.2"
8
+ description = "Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
11
+ keywords = ["diffBloch", "electron diffraction", "electron crystallography", "structure refinement", "PyTorch acceleration", "Metal GPU"]
11
12
  license = "MIT"
12
13
  authors = [
13
14
  { name = "Abhishek Shivakumar", email = "abhishek@example.com" }
@@ -16,6 +17,9 @@ classifiers = [
16
17
  "Development Status :: 4 - Beta",
17
18
  "Intended Audience :: Science/Research",
18
19
  "License :: OSI Approved :: MIT License",
20
+ "Operating System :: MacOS",
21
+ "Operating System :: MacOS :: MacOS X",
22
+ "Operating System :: POSIX :: Linux",
19
23
  "Programming Language :: Python :: 3",
20
24
  "Programming Language :: Python :: 3.10",
21
25
  "Programming Language :: Python :: 3.11",
@@ -38,8 +42,11 @@ diffbloch-fast = "faster_diffbloch.cli:main"
38
42
 
39
43
  [project.urls]
40
44
  Homepage = "https://godofecht.github.io/diffFlow/"
45
+ Documentation = "https://godofecht.github.io/diffFlow/"
41
46
  Repository = "https://github.com/godofecht/diffFlow"
42
47
  Issues = "https://github.com/godofecht/diffFlow/issues"
48
+ PyPI = "https://pypi.org/project/faster-diffbloch/"
49
+ piwheels = "https://www.piwheels.org/project/faster-diffbloch/"
43
50
  "Original diffBloch" = "https://diffbloch.com"
44
51
 
45
52
  [tool.hatch.build.targets.wheel]
@@ -4,7 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  from .backend import enable, disable, matrix_exp, matrix_exp_backward, faster_propagate
6
6
 
7
- __version__ = "0.1.0"
7
+ __version__ = "0.1.2"
8
8
  __all__ = [
9
9
  "enable",
10
10
  "disable",
@@ -3,17 +3,25 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import ctypes
6
+ import warnings
6
7
  from typing import Literal
7
8
  import numpy as np
8
9
  import torch
9
10
 
10
- from .builder import build_and_load_library
11
+ from .builder import (
12
+ build_and_load_library,
13
+ has_metal_support,
14
+ is_apple_silicon,
15
+ is_linux,
16
+ is_macos,
17
+ is_windows,
18
+ )
11
19
 
12
20
  _LIB: ctypes.CDLL | None = None
13
21
  _CURRENT_DEVICE: Literal["cpu", "gpu"] = "gpu"
14
22
 
15
23
 
16
- def get_library() -> ctypes.CDLL:
24
+ def get_library() -> ctypes.CDLL | None:
17
25
  global _LIB
18
26
  if _LIB is None:
19
27
  _LIB = build_and_load_library()
@@ -22,6 +30,11 @@ def get_library() -> ctypes.CDLL:
22
30
 
23
31
  def matrix_exp(a: np.ndarray, device: str = "gpu") -> np.ndarray:
24
32
  lib = get_library()
33
+ if lib is None:
34
+ # Fallback to PyTorch ATen
35
+ t = torch.from_numpy(a)
36
+ return torch.matrix_exp(t).numpy()
37
+
25
38
  a_c = np.ascontiguousarray(a, dtype=np.complex64)
26
39
  out = np.empty_like(a_c)
27
40
  n = a_c.shape[-1]
@@ -29,23 +42,35 @@ def matrix_exp(a: np.ndarray, device: str = "gpu") -> np.ndarray:
29
42
  ptr_in = a_c.ctypes.data_as(ctypes.c_void_p)
30
43
  ptr_out = out.ctypes.data_as(ctypes.c_void_p)
31
44
 
32
- if device == "gpu":
45
+ if device == "gpu" and is_macos():
33
46
  fn = getattr(lib, "bridge_matrix_exp_gpu_ptr_c64_i32_i32_ptr_c64", None)
34
47
  if fn:
35
48
  fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
36
- fn.restype = None
37
- fn(ptr_in, batch, n, ptr_out)
38
- return out
49
+ fn.restype = ctypes.c_int32
50
+ if fn(ptr_in, n, batch, ptr_out) == 0:
51
+ return out
52
+
53
+ fn = getattr(lib, "bridge_matrix_exp_ptr_c64_i32_i32_ptr_c64", None)
54
+ if fn:
55
+ fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
56
+ fn.restype = None
57
+ fn(ptr_in, n, batch, ptr_out)
58
+ return out
39
59
 
40
- fn = getattr(lib, "bridge_matrix_exp_ptr_c64_i32_i32_ptr_c64")
41
- fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
42
- fn.restype = None
43
- fn(ptr_in, batch, n, ptr_out)
44
- return out
60
+ return torch.matrix_exp(torch.from_numpy(a)).numpy()
45
61
 
46
62
 
47
63
  def matrix_exp_backward(a: np.ndarray, ebar: np.ndarray, dense: bool = False, device: str = "gpu") -> np.ndarray:
48
64
  lib = get_library()
65
+ if lib is None:
66
+ # Custom autograd.Function backward methods run with grad mode disabled.
67
+ # Re-enable it for the correctness fallback.
68
+ with torch.enable_grad():
69
+ tA = torch.from_numpy(a).requires_grad_(True)
70
+ tE = torch.matrix_exp(tA)
71
+ tE.backward(torch.from_numpy(ebar))
72
+ return tA.grad.numpy()
73
+
49
74
  a_c = np.ascontiguousarray(a, dtype=np.complex64)
50
75
  ebar_c = np.ascontiguousarray(ebar, dtype=np.complex64)
51
76
  out = np.empty_like(a_c)
@@ -57,19 +82,25 @@ def matrix_exp_backward(a: np.ndarray, ebar: np.ndarray, dense: bool = False, de
57
82
  ptr_ebar = ebar_c.ctypes.data_as(ctypes.c_void_p)
58
83
  ptr_out = out.ctypes.data_as(ctypes.c_void_p)
59
84
 
60
- if device == "gpu":
85
+ if device == "gpu" and is_macos():
61
86
  fn = getattr(lib, "bridge_matrix_exp_backward_gpu_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64", None)
62
87
  if fn:
63
88
  fn.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
64
- fn.restype = None
65
- fn(ptr_a, ptr_ebar, dense_val, batch, n, ptr_out)
66
- return out
89
+ fn.restype = ctypes.c_int32
90
+ if fn(ptr_a, ptr_ebar, n, batch, dense_val, ptr_out) == 0:
91
+ return out
67
92
 
68
- fn = getattr(lib, "bridge_matrix_exp_backward_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64")
69
- fn.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
70
- fn.restype = None
71
- fn(ptr_a, ptr_ebar, dense_val, batch, n, ptr_out)
72
- return out
93
+ fn = getattr(lib, "bridge_matrix_exp_backward_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64", None)
94
+ if fn:
95
+ fn.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
96
+ fn.restype = None
97
+ fn(ptr_a, ptr_ebar, n, batch, dense_val, ptr_out)
98
+ return out
99
+
100
+ tA = torch.from_numpy(a).requires_grad_(True)
101
+ tE = torch.matrix_exp(tA)
102
+ tE.backward(torch.from_numpy(ebar))
103
+ return tA.grad.numpy()
73
104
 
74
105
 
75
106
  class FasterMatrixExp(torch.autograd.Function):
@@ -126,8 +157,26 @@ def faster_propagate(system, thicknesses, *, method="matrix_exp", max_batch=None
126
157
 
127
158
 
128
159
  def enable(device: Literal["cpu", "gpu"] = "gpu") -> None:
129
- """Inject faster-diffbloch acceleration into diffBloch runtime."""
160
+ """Inject faster-diffbloch acceleration into diffBloch runtime with platform checking."""
130
161
  global _CURRENT_DEVICE
162
+
163
+ if device == "gpu":
164
+ if not is_macos():
165
+ warnings.warn(
166
+ "Metal GPU acceleration is available on macOS only. "
167
+ "Automatically selecting device='cpu' on this platform.",
168
+ UserWarning,
169
+ stacklevel=2,
170
+ )
171
+ device = "cpu"
172
+ elif not is_apple_silicon():
173
+ warnings.warn(
174
+ "Metal GPU acceleration is optimized for Apple Silicon unified memory. "
175
+ "Running on Intel macOS.",
176
+ UserWarning,
177
+ stacklevel=2,
178
+ )
179
+
131
180
  _CURRENT_DEVICE = device
132
181
  import importlib
133
182
  try:
@@ -0,0 +1,122 @@
1
+ """Native runtime builder, platform guards, and loader for faster-diffbloch."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ctypes
6
+ import os
7
+ import platform
8
+ import shutil
9
+ import subprocess
10
+ import sys
11
+ import warnings
12
+ from pathlib import Path
13
+
14
+ NATIVE_DIR = Path(__file__).resolve().parent / "native"
15
+ BUILD_DIR = Path.home() / ".cache" / "faster_diffbloch"
16
+
17
+
18
+ def is_macos() -> bool:
19
+ return sys.platform == "darwin"
20
+
21
+
22
+ def is_apple_silicon() -> bool:
23
+ return is_macos() and platform.machine() in ("arm64", "aarch64")
24
+
25
+
26
+ def is_linux() -> bool:
27
+ return sys.platform.startswith("linux")
28
+
29
+
30
+ def is_windows() -> bool:
31
+ return sys.platform.startswith("win") or sys.platform == "cygwin"
32
+
33
+
34
+ def has_metal_support() -> bool:
35
+ if not is_macos():
36
+ return False
37
+ # Check if xcrun and metal compiler are available
38
+ return shutil.which("xcrun") is not None
39
+
40
+
41
+ def get_c_compiler() -> str | None:
42
+ for cc in ("clang", "gcc", "cc"):
43
+ if shutil.which(cc) is not None:
44
+ return cc
45
+ return None
46
+
47
+
48
+ def build_and_load_library() -> ctypes.CDLL | None:
49
+ """Build or load the cached native acceleration library with platform guards."""
50
+ if is_windows():
51
+ warnings.warn(
52
+ "faster-diffbloch native acceleration is not currently supported on Windows. "
53
+ "Falling back to standard PyTorch execution.",
54
+ RuntimeWarning,
55
+ stacklevel=2,
56
+ )
57
+ return None
58
+
59
+ compiler = get_c_compiler()
60
+ if compiler is None:
61
+ warnings.warn(
62
+ "No C compiler (clang/gcc) found. "
63
+ "Install clang or build-essential to enable faster-diffbloch acceleration. "
64
+ "Falling back to standard PyTorch execution.",
65
+ RuntimeWarning,
66
+ stacklevel=2,
67
+ )
68
+ return None
69
+
70
+ BUILD_DIR.mkdir(parents=True, exist_ok=True)
71
+ lib_name = "libfaster_diffbloch.dylib" if is_macos() else "libfaster_diffbloch.so"
72
+ lib_path = BUILD_DIR / lib_name
73
+
74
+ sources = [
75
+ NATIVE_DIR / "bridge_lib.c",
76
+ NATIVE_DIR / "batch_cgemm.c",
77
+ NATIVE_DIR / "native_scattering.c",
78
+ ]
79
+
80
+ if is_macos():
81
+ sources.append(NATIVE_DIR / "metal_batch_cgemm.m")
82
+ if has_metal_support():
83
+ metal_src = NATIVE_DIR / "batch_cgemm.metal"
84
+ metallib_path = BUILD_DIR / "batch_cgemm.metallib"
85
+ if metal_src.exists() and (not metallib_path.exists() or metallib_path.stat().st_mtime < metal_src.stat().st_mtime):
86
+ try:
87
+ subprocess.run(
88
+ ["xcrun", "-sdk", "macosx", "metal", "-O3", "-c", str(metal_src), "-o", str(BUILD_DIR / "batch_cgemm.air")],
89
+ check=True, capture_output=True,
90
+ )
91
+ subprocess.run(
92
+ ["xcrun", "-sdk", "macosx", "metallib", str(BUILD_DIR / "batch_cgemm.air"), "-o", str(metallib_path)],
93
+ check=True, capture_output=True,
94
+ )
95
+ except Exception as exc:
96
+ warnings.warn(f"Metal shader precompilation skipped: {exc}", RuntimeWarning, stacklevel=2)
97
+
98
+ stale = not lib_path.exists() or any(lib_path.stat().st_mtime < s.stat().st_mtime for s in sources if s.exists())
99
+ if stale:
100
+ frameworks = ["-framework", "Accelerate", "-framework", "Metal", "-framework", "Foundation"] if is_macos() else ["-lblas"]
101
+ cmd = [
102
+ compiler, "-std=c11", "-O3", "-fPIC", "-shared",
103
+ "-D_DEFAULT_SOURCE", "-Wno-unused-function", "-Wno-unused-variable",
104
+ f"-I{NATIVE_DIR}",
105
+ *[str(s) for s in sources if s.exists()],
106
+ *frameworks, "-lm", "-o", str(lib_path),
107
+ ]
108
+ res = subprocess.run(cmd, capture_output=True, text=True)
109
+ if res.returncode != 0:
110
+ warnings.warn(
111
+ f"Failed to build faster-diffbloch native library with {compiler}:\n{res.stderr}\n"
112
+ "Falling back to standard PyTorch execution.",
113
+ RuntimeWarning,
114
+ stacklevel=2,
115
+ )
116
+ return None
117
+
118
+ try:
119
+ return ctypes.CDLL(str(lib_path))
120
+ except Exception as exc:
121
+ warnings.warn(f"Could not load faster-diffbloch library: {exc}", RuntimeWarning, stacklevel=2)
122
+ return None
@@ -1,98 +0,0 @@
1
- Metadata-Version: 2.5
2
- Name: faster-diffbloch
3
- Version: 0.1.0
4
- Summary: Drop-in Metal GPU and CPU acceleration for diffBloch electron crystallography
5
- Project-URL: Homepage, https://godofecht.github.io/diffFlow/
6
- Project-URL: Repository, https://github.com/godofecht/diffFlow
7
- Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
8
- Project-URL: Original diffBloch, https://diffbloch.com
9
- Author-email: Abhishek Shivakumar <abhishek@example.com>
10
- License-Expression: MIT
11
- License-File: LICENSE
12
- Classifier: Development Status :: 4 - Beta
13
- Classifier: Intended Audience :: Science/Research
14
- Classifier: License :: OSI Approved :: MIT License
15
- Classifier: Programming Language :: Python :: 3
16
- Classifier: Programming Language :: Python :: 3.10
17
- Classifier: Programming Language :: Python :: 3.11
18
- Classifier: Programming Language :: Python :: 3.12
19
- Classifier: Topic :: Scientific/Engineering :: Physics
20
- Requires-Python: >=3.10
21
- Requires-Dist: numpy>=1.24
22
- Requires-Dist: torch>=2.0
23
- Provides-Extra: diffbloch
24
- Requires-Dist: diffbloch; extra == 'diffbloch'
25
- Description-Content-Type: text/markdown
26
-
27
- # faster-diffBloch
28
-
29
- Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
30
-
31
- Documentation and comparison benchmarks: [https://godofecht.github.io/diffFlow/](https://godofecht.github.io/diffFlow/)
32
-
33
- Original diffBloch project: [https://diffbloch.com](https://diffbloch.com)
34
-
35
- ---
36
-
37
- ## Why faster-diffBloch?
38
-
39
- 1. **Native Metal GPU Execution:**
40
- PyTorch MPS lacks a native GPU kernel for `aten::linalg_matrix_exp`, which causes PyTorch to fall back to CPU execution with host-device memory transfers. `faster-diffBloch` executes matrix exponentials directly on Apple Silicon Metal with zero-copy unified memory.
41
-
42
- 2. **Blocked-Pair Adjoint Formulation:**
43
- Standard matrix exponential autograd embeds the operator into a $2N \times 2N$ block matrix, costing $8 \times N^3$ FLOPs. `faster-diffBloch` evaluates the pullback in the block-triangular pair algebra $(Y_a Y_b, Y_a L_b + L_a Y_b)$, reducing the work to $3 \times N^3$ FLOPs (2.67x fewer products).
44
-
45
- 3. **Bit-for-Bit Validation:**
46
- Passes all 738 unit tests in diffBloch and reproduces the experimental 99-rotation quartz dataset ($R_{\text{obs}} = 0.0486$).
47
-
48
- ---
49
-
50
- ## Performance
51
-
52
- Forward and backward timing comparison on Apple Silicon (M4 Max) at $N=579$ beams (CsPbBr3 scale):
53
-
54
- | Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
55
- | :--- | :---: | :---: | :---: | :---: |
56
- | PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
57
- | PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
58
- | **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
59
- | **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
60
-
61
- ---
62
-
63
- ## Installation
64
-
65
- ```bash
66
- pip install faster-diffbloch
67
- ```
68
-
69
- ---
70
-
71
- ## Usage
72
-
73
- ### 1. Drop-in CLI
74
-
75
- Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
76
-
77
- ```bash
78
- diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
79
- diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
80
- ```
81
-
82
- ### 2. Python API Injection
83
-
84
- Enable acceleration inside any existing diffBloch script:
85
-
86
- ```python
87
- import faster_diffbloch
88
-
89
- # Enable Metal GPU acceleration
90
- faster_diffbloch.enable(device="gpu")
91
-
92
- # Or CPU acceleration
93
- faster_diffbloch.enable(device="cpu")
94
-
95
- # Run standard diffBloch code
96
- import diffBloch
97
- # All propagate and matrix_exp calls now route through faster-diffBloch
98
- ```
@@ -1,72 +0,0 @@
1
- # faster-diffBloch
2
-
3
- Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
4
-
5
- Documentation and comparison benchmarks: [https://godofecht.github.io/diffFlow/](https://godofecht.github.io/diffFlow/)
6
-
7
- Original diffBloch project: [https://diffbloch.com](https://diffbloch.com)
8
-
9
- ---
10
-
11
- ## Why faster-diffBloch?
12
-
13
- 1. **Native Metal GPU Execution:**
14
- PyTorch MPS lacks a native GPU kernel for `aten::linalg_matrix_exp`, which causes PyTorch to fall back to CPU execution with host-device memory transfers. `faster-diffBloch` executes matrix exponentials directly on Apple Silicon Metal with zero-copy unified memory.
15
-
16
- 2. **Blocked-Pair Adjoint Formulation:**
17
- Standard matrix exponential autograd embeds the operator into a $2N \times 2N$ block matrix, costing $8 \times N^3$ FLOPs. `faster-diffBloch` evaluates the pullback in the block-triangular pair algebra $(Y_a Y_b, Y_a L_b + L_a Y_b)$, reducing the work to $3 \times N^3$ FLOPs (2.67x fewer products).
18
-
19
- 3. **Bit-for-Bit Validation:**
20
- Passes all 738 unit tests in diffBloch and reproduces the experimental 99-rotation quartz dataset ($R_{\text{obs}} = 0.0486$).
21
-
22
- ---
23
-
24
- ## Performance
25
-
26
- Forward and backward timing comparison on Apple Silicon (M4 Max) at $N=579$ beams (CsPbBr3 scale):
27
-
28
- | Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
29
- | :--- | :---: | :---: | :---: | :---: |
30
- | PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
31
- | PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
32
- | **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
33
- | **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
34
-
35
- ---
36
-
37
- ## Installation
38
-
39
- ```bash
40
- pip install faster-diffbloch
41
- ```
42
-
43
- ---
44
-
45
- ## Usage
46
-
47
- ### 1. Drop-in CLI
48
-
49
- Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
50
-
51
- ```bash
52
- diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
53
- diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
54
- ```
55
-
56
- ### 2. Python API Injection
57
-
58
- Enable acceleration inside any existing diffBloch script:
59
-
60
- ```python
61
- import faster_diffbloch
62
-
63
- # Enable Metal GPU acceleration
64
- faster_diffbloch.enable(device="gpu")
65
-
66
- # Or CPU acceleration
67
- faster_diffbloch.enable(device="cpu")
68
-
69
- # Run standard diffBloch code
70
- import diffBloch
71
- # All propagate and matrix_exp calls now route through faster-diffBloch
72
- ```
@@ -1,58 +0,0 @@
1
- """Native runtime builder and loader for faster-diffbloch."""
2
-
3
- from __future__ import annotations
4
-
5
- import ctypes
6
- import os
7
- import subprocess
8
- import sys
9
- from pathlib import Path
10
-
11
- NATIVE_DIR = Path(__file__).resolve().parent / "native"
12
- BUILD_DIR = Path.home() / ".cache" / "faster_diffbloch"
13
-
14
-
15
- def build_and_load_library() -> ctypes.CDLL:
16
- """Build or load the cached native acceleration library."""
17
- BUILD_DIR.mkdir(parents=True, exist_ok=True)
18
- lib_name = "libfaster_diffbloch.dylib" if sys.platform == "darwin" else "libfaster_diffbloch.so"
19
- lib_path = BUILD_DIR / lib_name
20
-
21
- sources = [
22
- NATIVE_DIR / "bridge_lib.c",
23
- NATIVE_DIR / "batch_cgemm.c",
24
- NATIVE_DIR / "native_scattering.c",
25
- ]
26
- if sys.platform == "darwin":
27
- sources.append(NATIVE_DIR / "metal_batch_cgemm.m")
28
- # Build metallib if metal is available
29
- metal_src = NATIVE_DIR / "batch_cgemm.metal"
30
- metallib_path = BUILD_DIR / "batch_cgemm.metallib"
31
- if metal_src.exists() and (not metallib_path.exists() or metallib_path.stat().st_mtime < metal_src.stat().st_mtime):
32
- try:
33
- subprocess.run(
34
- ["xcrun", "-sdk", "macosx", "metal", "-O3", "-c", str(metal_src), "-o", str(BUILD_DIR / "batch_cgemm.air")],
35
- check=True, capture_output=True
36
- )
37
- subprocess.run(
38
- ["xcrun", "-sdk", "macosx", "metallib", str(BUILD_DIR / "batch_cgemm.air"), "-o", str(metallib_path)],
39
- check=True, capture_output=True
40
- )
41
- except Exception:
42
- pass
43
-
44
- stale = not lib_path.exists() or any(lib_path.stat().st_mtime < s.stat().st_mtime for s in sources if s.exists())
45
- if stale:
46
- frameworks = ["-framework", "Accelerate", "-framework", "Metal", "-framework", "Foundation"] if sys.platform == "darwin" else ["-lblas"]
47
- cmd = [
48
- "clang", "-std=c11", "-O3", "-fPIC", "-shared",
49
- "-D_DEFAULT_SOURCE", "-Wno-unused-function", "-Wno-unused-variable",
50
- f"-I{NATIVE_DIR}",
51
- *[str(s) for s in sources if s.exists()],
52
- *frameworks, "-lm", "-o", str(lib_path),
53
- ]
54
- res = subprocess.run(cmd, capture_output=True, text=True)
55
- if res.returncode != 0:
56
- raise RuntimeError(f"Failed to build faster-diffbloch native library:\n{res.stderr}")
57
-
58
- return ctypes.CDLL(str(lib_path))