faster-diffbloch 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- faster_diffbloch-0.1.2/PKG-INFO +145 -0
- faster_diffbloch-0.1.2/README.md +112 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/pyproject.toml +9 -2
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/__init__.py +1 -1
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/backend.py +70 -21
- faster_diffbloch-0.1.2/src/faster_diffbloch/builder.py +122 -0
- faster_diffbloch-0.1.0/PKG-INFO +0 -98
- faster_diffbloch-0.1.0/README.md +0 -72
- faster_diffbloch-0.1.0/src/faster_diffbloch/builder.py +0 -58
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/.gitignore +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/LICENSE +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/cli.py +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.c +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.h +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.metal +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/bridge_lib.c +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/metal_batch_cgemm.m +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.c +0 -0
- {faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.h +0 -0
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: faster-diffbloch
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch
|
|
5
|
+
Project-URL: Homepage, https://godofecht.github.io/diffFlow/
|
|
6
|
+
Project-URL: Documentation, https://godofecht.github.io/diffFlow/
|
|
7
|
+
Project-URL: Repository, https://github.com/godofecht/diffFlow
|
|
8
|
+
Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
|
|
9
|
+
Project-URL: PyPI, https://pypi.org/project/faster-diffbloch/
|
|
10
|
+
Project-URL: piwheels, https://www.piwheels.org/project/faster-diffbloch/
|
|
11
|
+
Project-URL: Original diffBloch, https://diffbloch.com
|
|
12
|
+
Author-email: Abhishek Shivakumar <abhishek@example.com>
|
|
13
|
+
License-Expression: MIT
|
|
14
|
+
License-File: LICENSE
|
|
15
|
+
Keywords: Metal GPU,PyTorch acceleration,diffBloch,electron crystallography,electron diffraction,structure refinement
|
|
16
|
+
Classifier: Development Status :: 4 - Beta
|
|
17
|
+
Classifier: Intended Audience :: Science/Research
|
|
18
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
19
|
+
Classifier: Operating System :: MacOS
|
|
20
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
21
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
22
|
+
Classifier: Programming Language :: Python :: 3
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
25
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
26
|
+
Classifier: Topic :: Scientific/Engineering :: Physics
|
|
27
|
+
Requires-Python: >=3.10
|
|
28
|
+
Requires-Dist: numpy>=1.24
|
|
29
|
+
Requires-Dist: torch>=2.0
|
|
30
|
+
Provides-Extra: diffbloch
|
|
31
|
+
Requires-Dist: diffbloch; extra == 'diffbloch'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
# faster-diffbloch
|
|
35
|
+
|
|
36
|
+
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
37
|
+
|
|
38
|
+
**Up to 1.80× faster than PyTorch CPU** in measured end-to-end forward-plus-backward benchmarks, with a Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux.
|
|
39
|
+
|
|
40
|
+
| Package | Documentation | Source repository | Original project |
|
|
41
|
+
| :--- | :--- | :--- | :--- |
|
|
42
|
+
| [PyPI](https://pypi.org/project/faster-diffbloch/) · [piwheels](https://www.piwheels.org/project/faster-diffbloch/) | [diffFlow documentation and benchmarks](https://godofecht.github.io/diffFlow/) | [github.com/godofecht/diffFlow](https://github.com/godofecht/diffFlow) | [diffbloch.com](https://diffbloch.com) |
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## Operating System and Platform Support
|
|
47
|
+
|
|
48
|
+
| Operating System / Hardware | CPU Acceleration (`device="cpu"`) | GPU Acceleration (`device="gpu"`) | Backend Runtime |
|
|
49
|
+
| :--- | :---: | :---: | :--- |
|
|
50
|
+
| **macOS Apple Silicon (M1/M2/M3/M4/Max/Ultra)** | **Supported** | **Supported** | Native Metal Compute Shaders + Apple Accelerate BLAS |
|
|
51
|
+
| **macOS Intel (x86_64)** | **Supported** | Fallback to CPU | Apple Accelerate BLAS |
|
|
52
|
+
| **Linux (x86_64 / aarch64)** | **Supported** | Fallback to CPU | OpenBLAS / C11 BLAS |
|
|
53
|
+
| **Windows** | PyTorch Reference | PyTorch Reference | Pure PyTorch Reference Fallback |
|
|
54
|
+
|
|
55
|
+
Runtime platform guards automatically detect your operating system and hardware configuration. When `device="gpu"` is requested on Linux, `faster-diffbloch` automatically selects the optimized CPU backend with an informative warning.
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## Why faster-diffBloch?
|
|
60
|
+
|
|
61
|
+
`faster-diffbloch` keeps the diffBloch workflow and public API intact while moving
|
|
62
|
+
the expensive propagation and gradient work onto an accelerated native backend.
|
|
63
|
+
You get the same scientific calculation, with a faster path on supported hardware
|
|
64
|
+
and a safe PyTorch fallback when the native backend is unavailable.
|
|
65
|
+
|
|
66
|
+
The package is validated against the diffBloch test suite and reference results.
|
|
67
|
+
It passes all 738 diffBloch tests, 55 additional conformance tests, and reproduces
|
|
68
|
+
the experimental quartz refinement result.
|
|
69
|
+
|
|
70
|
+
---
|
|
71
|
+
|
|
72
|
+
## Performance
|
|
73
|
+
|
|
74
|
+
End-to-end forward-plus-backward timing on an Apple M4 Max, measured across the
|
|
75
|
+
same diffBloch workload at different beam counts:
|
|
76
|
+
|
|
77
|
+
| Beams | faster-diffBloch | PyTorch | Speedup |
|
|
78
|
+
| :---: | :---: | :---: | :---: |
|
|
79
|
+
| 31 | 0.90 ms | 1.62 ms | **1.80×** |
|
|
80
|
+
| 61 | 1.37 ms | 2.24 ms | **1.63×** |
|
|
81
|
+
| 91 | 2.84 ms | 3.99 ms | **1.41×** |
|
|
82
|
+
| 163 | 7.67 ms | 10.98 ms | **1.43×** |
|
|
83
|
+
| 579 | 89.81 ms | 135.77 ms | **1.51×** |
|
|
84
|
+
|
|
85
|
+
That is **1.41×–1.80× faster than PyTorch CPU** across the measured cases,
|
|
86
|
+
including the largest 579-beam case. The same benchmark is **1.21× faster than
|
|
87
|
+
the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
|
|
88
|
+
essentially level at 579 beams.
|
|
89
|
+
|
|
90
|
+
On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
|
|
91
|
+
comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
|
|
92
|
+
forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
|
|
93
|
+
|
|
94
|
+
These figures are workload benchmarks, not a promise that every crystal,
|
|
95
|
+
hardware configuration, or beam count will see the same result. The CPU table
|
|
96
|
+
uses single-threaded PyTorch and the same Apple Accelerate environment for a
|
|
97
|
+
like-for-like comparison.
|
|
98
|
+
|
|
99
|
+
### Package benchmark snapshot
|
|
100
|
+
|
|
101
|
+
| Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
|
|
102
|
+
| :--- | :---: | :---: | :---: | :---: |
|
|
103
|
+
| PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
|
|
104
|
+
| PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
|
|
105
|
+
| **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
|
|
106
|
+
| **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Installation
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
pip install faster-diffbloch
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
---
|
|
117
|
+
|
|
118
|
+
## Usage
|
|
119
|
+
|
|
120
|
+
### 1. Drop-in CLI
|
|
121
|
+
|
|
122
|
+
Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
126
|
+
diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
### 2. Python API Injection
|
|
130
|
+
|
|
131
|
+
Enable acceleration inside any existing diffBloch script:
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
import faster_diffbloch
|
|
135
|
+
|
|
136
|
+
# Enable Metal GPU acceleration (macOS Apple Silicon)
|
|
137
|
+
faster_diffbloch.enable(device="gpu")
|
|
138
|
+
|
|
139
|
+
# Or CPU acceleration (macOS and Linux)
|
|
140
|
+
faster_diffbloch.enable(device="cpu")
|
|
141
|
+
|
|
142
|
+
# Run standard diffBloch code
|
|
143
|
+
import diffBloch
|
|
144
|
+
# All propagate and matrix_exp calls now route through faster-diffBloch
|
|
145
|
+
```
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# faster-diffbloch
|
|
2
|
+
|
|
3
|
+
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
4
|
+
|
|
5
|
+
**Up to 1.80× faster than PyTorch CPU** in measured end-to-end forward-plus-backward benchmarks, with a Metal GPU path for Apple Silicon and an optimized CPU path for macOS and Linux.
|
|
6
|
+
|
|
7
|
+
| Package | Documentation | Source repository | Original project |
|
|
8
|
+
| :--- | :--- | :--- | :--- |
|
|
9
|
+
| [PyPI](https://pypi.org/project/faster-diffbloch/) · [piwheels](https://www.piwheels.org/project/faster-diffbloch/) | [diffFlow documentation and benchmarks](https://godofecht.github.io/diffFlow/) | [github.com/godofecht/diffFlow](https://github.com/godofecht/diffFlow) | [diffbloch.com](https://diffbloch.com) |
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## Operating System and Platform Support
|
|
14
|
+
|
|
15
|
+
| Operating System / Hardware | CPU Acceleration (`device="cpu"`) | GPU Acceleration (`device="gpu"`) | Backend Runtime |
|
|
16
|
+
| :--- | :---: | :---: | :--- |
|
|
17
|
+
| **macOS Apple Silicon (M1/M2/M3/M4/Max/Ultra)** | **Supported** | **Supported** | Native Metal Compute Shaders + Apple Accelerate BLAS |
|
|
18
|
+
| **macOS Intel (x86_64)** | **Supported** | Fallback to CPU | Apple Accelerate BLAS |
|
|
19
|
+
| **Linux (x86_64 / aarch64)** | **Supported** | Fallback to CPU | OpenBLAS / C11 BLAS |
|
|
20
|
+
| **Windows** | PyTorch Reference | PyTorch Reference | Pure PyTorch Reference Fallback |
|
|
21
|
+
|
|
22
|
+
Runtime platform guards automatically detect your operating system and hardware configuration. When `device="gpu"` is requested on Linux, `faster-diffbloch` automatically selects the optimized CPU backend with an informative warning.
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## Why faster-diffBloch?
|
|
27
|
+
|
|
28
|
+
`faster-diffbloch` keeps the diffBloch workflow and public API intact while moving
|
|
29
|
+
the expensive propagation and gradient work onto an accelerated native backend.
|
|
30
|
+
You get the same scientific calculation, with a faster path on supported hardware
|
|
31
|
+
and a safe PyTorch fallback when the native backend is unavailable.
|
|
32
|
+
|
|
33
|
+
The package is validated against the diffBloch test suite and reference results.
|
|
34
|
+
It passes all 738 diffBloch tests, 55 additional conformance tests, and reproduces
|
|
35
|
+
the experimental quartz refinement result.
|
|
36
|
+
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
## Performance
|
|
40
|
+
|
|
41
|
+
End-to-end forward-plus-backward timing on an Apple M4 Max, measured across the
|
|
42
|
+
same diffBloch workload at different beam counts:
|
|
43
|
+
|
|
44
|
+
| Beams | faster-diffBloch | PyTorch | Speedup |
|
|
45
|
+
| :---: | :---: | :---: | :---: |
|
|
46
|
+
| 31 | 0.90 ms | 1.62 ms | **1.80×** |
|
|
47
|
+
| 61 | 1.37 ms | 2.24 ms | **1.63×** |
|
|
48
|
+
| 91 | 2.84 ms | 3.99 ms | **1.41×** |
|
|
49
|
+
| 163 | 7.67 ms | 10.98 ms | **1.43×** |
|
|
50
|
+
| 579 | 89.81 ms | 135.77 ms | **1.51×** |
|
|
51
|
+
|
|
52
|
+
That is **1.41×–1.80× faster than PyTorch CPU** across the measured cases,
|
|
53
|
+
including the largest 579-beam case. The same benchmark is **1.21× faster than
|
|
54
|
+
the Mojo reference at 31 beams**, remains ahead through 163 beams, and is
|
|
55
|
+
essentially level at 579 beams.
|
|
56
|
+
|
|
57
|
+
On Apple Silicon, the Metal GPU path is also available. In the package's M4 Max
|
|
58
|
+
comparison run at 579 beams, it measured **2.24× faster than PyTorch CPU for the
|
|
59
|
+
forward pass** and **2.63× faster than PyTorch MPS for forward plus backward**.
|
|
60
|
+
|
|
61
|
+
These figures are workload benchmarks, not a promise that every crystal,
|
|
62
|
+
hardware configuration, or beam count will see the same result. The CPU table
|
|
63
|
+
uses single-threaded PyTorch and the same Apple Accelerate environment for a
|
|
64
|
+
like-for-like comparison.
|
|
65
|
+
|
|
66
|
+
### Package benchmark snapshot
|
|
67
|
+
|
|
68
|
+
| Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
|
|
69
|
+
| :--- | :---: | :---: | :---: | :---: |
|
|
70
|
+
| PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
|
|
71
|
+
| PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
|
|
72
|
+
| **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
|
|
73
|
+
| **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
|
|
74
|
+
|
|
75
|
+
---
|
|
76
|
+
|
|
77
|
+
## Installation
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install faster-diffbloch
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
---
|
|
84
|
+
|
|
85
|
+
## Usage
|
|
86
|
+
|
|
87
|
+
### 1. Drop-in CLI
|
|
88
|
+
|
|
89
|
+
Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
93
|
+
diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
### 2. Python API Injection
|
|
97
|
+
|
|
98
|
+
Enable acceleration inside any existing diffBloch script:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
import faster_diffbloch
|
|
102
|
+
|
|
103
|
+
# Enable Metal GPU acceleration (macOS Apple Silicon)
|
|
104
|
+
faster_diffbloch.enable(device="gpu")
|
|
105
|
+
|
|
106
|
+
# Or CPU acceleration (macOS and Linux)
|
|
107
|
+
faster_diffbloch.enable(device="cpu")
|
|
108
|
+
|
|
109
|
+
# Run standard diffBloch code
|
|
110
|
+
import diffBloch
|
|
111
|
+
# All propagate and matrix_exp calls now route through faster-diffBloch
|
|
112
|
+
```
|
|
@@ -4,10 +4,11 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "faster-diffbloch"
|
|
7
|
-
version = "0.1.
|
|
8
|
-
description = "Drop-in Metal GPU and CPU acceleration for diffBloch
|
|
7
|
+
version = "0.1.2"
|
|
8
|
+
description = "Drop-in Metal GPU (macOS) and optimized CPU (macOS/Linux) acceleration for diffBloch"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
|
+
keywords = ["diffBloch", "electron diffraction", "electron crystallography", "structure refinement", "PyTorch acceleration", "Metal GPU"]
|
|
11
12
|
license = "MIT"
|
|
12
13
|
authors = [
|
|
13
14
|
{ name = "Abhishek Shivakumar", email = "abhishek@example.com" }
|
|
@@ -16,6 +17,9 @@ classifiers = [
|
|
|
16
17
|
"Development Status :: 4 - Beta",
|
|
17
18
|
"Intended Audience :: Science/Research",
|
|
18
19
|
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Operating System :: MacOS",
|
|
21
|
+
"Operating System :: MacOS :: MacOS X",
|
|
22
|
+
"Operating System :: POSIX :: Linux",
|
|
19
23
|
"Programming Language :: Python :: 3",
|
|
20
24
|
"Programming Language :: Python :: 3.10",
|
|
21
25
|
"Programming Language :: Python :: 3.11",
|
|
@@ -38,8 +42,11 @@ diffbloch-fast = "faster_diffbloch.cli:main"
|
|
|
38
42
|
|
|
39
43
|
[project.urls]
|
|
40
44
|
Homepage = "https://godofecht.github.io/diffFlow/"
|
|
45
|
+
Documentation = "https://godofecht.github.io/diffFlow/"
|
|
41
46
|
Repository = "https://github.com/godofecht/diffFlow"
|
|
42
47
|
Issues = "https://github.com/godofecht/diffFlow/issues"
|
|
48
|
+
PyPI = "https://pypi.org/project/faster-diffbloch/"
|
|
49
|
+
piwheels = "https://www.piwheels.org/project/faster-diffbloch/"
|
|
43
50
|
"Original diffBloch" = "https://diffbloch.com"
|
|
44
51
|
|
|
45
52
|
[tool.hatch.build.targets.wheel]
|
|
@@ -3,17 +3,25 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import ctypes
|
|
6
|
+
import warnings
|
|
6
7
|
from typing import Literal
|
|
7
8
|
import numpy as np
|
|
8
9
|
import torch
|
|
9
10
|
|
|
10
|
-
from .builder import
|
|
11
|
+
from .builder import (
|
|
12
|
+
build_and_load_library,
|
|
13
|
+
has_metal_support,
|
|
14
|
+
is_apple_silicon,
|
|
15
|
+
is_linux,
|
|
16
|
+
is_macos,
|
|
17
|
+
is_windows,
|
|
18
|
+
)
|
|
11
19
|
|
|
12
20
|
_LIB: ctypes.CDLL | None = None
|
|
13
21
|
_CURRENT_DEVICE: Literal["cpu", "gpu"] = "gpu"
|
|
14
22
|
|
|
15
23
|
|
|
16
|
-
def get_library() -> ctypes.CDLL:
|
|
24
|
+
def get_library() -> ctypes.CDLL | None:
|
|
17
25
|
global _LIB
|
|
18
26
|
if _LIB is None:
|
|
19
27
|
_LIB = build_and_load_library()
|
|
@@ -22,6 +30,11 @@ def get_library() -> ctypes.CDLL:
|
|
|
22
30
|
|
|
23
31
|
def matrix_exp(a: np.ndarray, device: str = "gpu") -> np.ndarray:
|
|
24
32
|
lib = get_library()
|
|
33
|
+
if lib is None:
|
|
34
|
+
# Fallback to PyTorch ATen
|
|
35
|
+
t = torch.from_numpy(a)
|
|
36
|
+
return torch.matrix_exp(t).numpy()
|
|
37
|
+
|
|
25
38
|
a_c = np.ascontiguousarray(a, dtype=np.complex64)
|
|
26
39
|
out = np.empty_like(a_c)
|
|
27
40
|
n = a_c.shape[-1]
|
|
@@ -29,23 +42,35 @@ def matrix_exp(a: np.ndarray, device: str = "gpu") -> np.ndarray:
|
|
|
29
42
|
ptr_in = a_c.ctypes.data_as(ctypes.c_void_p)
|
|
30
43
|
ptr_out = out.ctypes.data_as(ctypes.c_void_p)
|
|
31
44
|
|
|
32
|
-
if device == "gpu":
|
|
45
|
+
if device == "gpu" and is_macos():
|
|
33
46
|
fn = getattr(lib, "bridge_matrix_exp_gpu_ptr_c64_i32_i32_ptr_c64", None)
|
|
34
47
|
if fn:
|
|
35
48
|
fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
|
|
36
|
-
fn.restype =
|
|
37
|
-
fn(ptr_in,
|
|
38
|
-
|
|
49
|
+
fn.restype = ctypes.c_int32
|
|
50
|
+
if fn(ptr_in, n, batch, ptr_out) == 0:
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
fn = getattr(lib, "bridge_matrix_exp_ptr_c64_i32_i32_ptr_c64", None)
|
|
54
|
+
if fn:
|
|
55
|
+
fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
|
|
56
|
+
fn.restype = None
|
|
57
|
+
fn(ptr_in, n, batch, ptr_out)
|
|
58
|
+
return out
|
|
39
59
|
|
|
40
|
-
|
|
41
|
-
fn.argtypes = [ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
|
|
42
|
-
fn.restype = None
|
|
43
|
-
fn(ptr_in, batch, n, ptr_out)
|
|
44
|
-
return out
|
|
60
|
+
return torch.matrix_exp(torch.from_numpy(a)).numpy()
|
|
45
61
|
|
|
46
62
|
|
|
47
63
|
def matrix_exp_backward(a: np.ndarray, ebar: np.ndarray, dense: bool = False, device: str = "gpu") -> np.ndarray:
|
|
48
64
|
lib = get_library()
|
|
65
|
+
if lib is None:
|
|
66
|
+
# Custom autograd.Function backward methods run with grad mode disabled.
|
|
67
|
+
# Re-enable it for the correctness fallback.
|
|
68
|
+
with torch.enable_grad():
|
|
69
|
+
tA = torch.from_numpy(a).requires_grad_(True)
|
|
70
|
+
tE = torch.matrix_exp(tA)
|
|
71
|
+
tE.backward(torch.from_numpy(ebar))
|
|
72
|
+
return tA.grad.numpy()
|
|
73
|
+
|
|
49
74
|
a_c = np.ascontiguousarray(a, dtype=np.complex64)
|
|
50
75
|
ebar_c = np.ascontiguousarray(ebar, dtype=np.complex64)
|
|
51
76
|
out = np.empty_like(a_c)
|
|
@@ -57,19 +82,25 @@ def matrix_exp_backward(a: np.ndarray, ebar: np.ndarray, dense: bool = False, de
|
|
|
57
82
|
ptr_ebar = ebar_c.ctypes.data_as(ctypes.c_void_p)
|
|
58
83
|
ptr_out = out.ctypes.data_as(ctypes.c_void_p)
|
|
59
84
|
|
|
60
|
-
if device == "gpu":
|
|
85
|
+
if device == "gpu" and is_macos():
|
|
61
86
|
fn = getattr(lib, "bridge_matrix_exp_backward_gpu_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64", None)
|
|
62
87
|
if fn:
|
|
63
88
|
fn.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
|
|
64
|
-
fn.restype =
|
|
65
|
-
fn(ptr_a, ptr_ebar,
|
|
66
|
-
|
|
89
|
+
fn.restype = ctypes.c_int32
|
|
90
|
+
if fn(ptr_a, ptr_ebar, n, batch, dense_val, ptr_out) == 0:
|
|
91
|
+
return out
|
|
67
92
|
|
|
68
|
-
fn = getattr(lib, "bridge_matrix_exp_backward_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64")
|
|
69
|
-
fn
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
93
|
+
fn = getattr(lib, "bridge_matrix_exp_backward_ptr_c64_ptr_c64_i32_i32_i32_ptr_c64", None)
|
|
94
|
+
if fn:
|
|
95
|
+
fn.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_int32, ctypes.c_int32, ctypes.c_int32, ctypes.c_void_p]
|
|
96
|
+
fn.restype = None
|
|
97
|
+
fn(ptr_a, ptr_ebar, n, batch, dense_val, ptr_out)
|
|
98
|
+
return out
|
|
99
|
+
|
|
100
|
+
tA = torch.from_numpy(a).requires_grad_(True)
|
|
101
|
+
tE = torch.matrix_exp(tA)
|
|
102
|
+
tE.backward(torch.from_numpy(ebar))
|
|
103
|
+
return tA.grad.numpy()
|
|
73
104
|
|
|
74
105
|
|
|
75
106
|
class FasterMatrixExp(torch.autograd.Function):
|
|
@@ -126,8 +157,26 @@ def faster_propagate(system, thicknesses, *, method="matrix_exp", max_batch=None
|
|
|
126
157
|
|
|
127
158
|
|
|
128
159
|
def enable(device: Literal["cpu", "gpu"] = "gpu") -> None:
|
|
129
|
-
"""Inject faster-diffbloch acceleration into diffBloch runtime."""
|
|
160
|
+
"""Inject faster-diffbloch acceleration into diffBloch runtime with platform checking."""
|
|
130
161
|
global _CURRENT_DEVICE
|
|
162
|
+
|
|
163
|
+
if device == "gpu":
|
|
164
|
+
if not is_macos():
|
|
165
|
+
warnings.warn(
|
|
166
|
+
"Metal GPU acceleration is available on macOS only. "
|
|
167
|
+
"Automatically selecting device='cpu' on this platform.",
|
|
168
|
+
UserWarning,
|
|
169
|
+
stacklevel=2,
|
|
170
|
+
)
|
|
171
|
+
device = "cpu"
|
|
172
|
+
elif not is_apple_silicon():
|
|
173
|
+
warnings.warn(
|
|
174
|
+
"Metal GPU acceleration is optimized for Apple Silicon unified memory. "
|
|
175
|
+
"Running on Intel macOS.",
|
|
176
|
+
UserWarning,
|
|
177
|
+
stacklevel=2,
|
|
178
|
+
)
|
|
179
|
+
|
|
131
180
|
_CURRENT_DEVICE = device
|
|
132
181
|
import importlib
|
|
133
182
|
try:
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Native runtime builder, platform guards, and loader for faster-diffbloch."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import ctypes
|
|
6
|
+
import os
|
|
7
|
+
import platform
|
|
8
|
+
import shutil
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
import warnings
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
NATIVE_DIR = Path(__file__).resolve().parent / "native"
|
|
15
|
+
BUILD_DIR = Path.home() / ".cache" / "faster_diffbloch"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def is_macos() -> bool:
|
|
19
|
+
return sys.platform == "darwin"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_apple_silicon() -> bool:
|
|
23
|
+
return is_macos() and platform.machine() in ("arm64", "aarch64")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def is_linux() -> bool:
|
|
27
|
+
return sys.platform.startswith("linux")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def is_windows() -> bool:
|
|
31
|
+
return sys.platform.startswith("win") or sys.platform == "cygwin"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def has_metal_support() -> bool:
|
|
35
|
+
if not is_macos():
|
|
36
|
+
return False
|
|
37
|
+
# Check if xcrun and metal compiler are available
|
|
38
|
+
return shutil.which("xcrun") is not None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def get_c_compiler() -> str | None:
|
|
42
|
+
for cc in ("clang", "gcc", "cc"):
|
|
43
|
+
if shutil.which(cc) is not None:
|
|
44
|
+
return cc
|
|
45
|
+
return None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def build_and_load_library() -> ctypes.CDLL | None:
|
|
49
|
+
"""Build or load the cached native acceleration library with platform guards."""
|
|
50
|
+
if is_windows():
|
|
51
|
+
warnings.warn(
|
|
52
|
+
"faster-diffbloch native acceleration is not currently supported on Windows. "
|
|
53
|
+
"Falling back to standard PyTorch execution.",
|
|
54
|
+
RuntimeWarning,
|
|
55
|
+
stacklevel=2,
|
|
56
|
+
)
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
compiler = get_c_compiler()
|
|
60
|
+
if compiler is None:
|
|
61
|
+
warnings.warn(
|
|
62
|
+
"No C compiler (clang/gcc) found. "
|
|
63
|
+
"Install clang or build-essential to enable faster-diffbloch acceleration. "
|
|
64
|
+
"Falling back to standard PyTorch execution.",
|
|
65
|
+
RuntimeWarning,
|
|
66
|
+
stacklevel=2,
|
|
67
|
+
)
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
BUILD_DIR.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
lib_name = "libfaster_diffbloch.dylib" if is_macos() else "libfaster_diffbloch.so"
|
|
72
|
+
lib_path = BUILD_DIR / lib_name
|
|
73
|
+
|
|
74
|
+
sources = [
|
|
75
|
+
NATIVE_DIR / "bridge_lib.c",
|
|
76
|
+
NATIVE_DIR / "batch_cgemm.c",
|
|
77
|
+
NATIVE_DIR / "native_scattering.c",
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
if is_macos():
|
|
81
|
+
sources.append(NATIVE_DIR / "metal_batch_cgemm.m")
|
|
82
|
+
if has_metal_support():
|
|
83
|
+
metal_src = NATIVE_DIR / "batch_cgemm.metal"
|
|
84
|
+
metallib_path = BUILD_DIR / "batch_cgemm.metallib"
|
|
85
|
+
if metal_src.exists() and (not metallib_path.exists() or metallib_path.stat().st_mtime < metal_src.stat().st_mtime):
|
|
86
|
+
try:
|
|
87
|
+
subprocess.run(
|
|
88
|
+
["xcrun", "-sdk", "macosx", "metal", "-O3", "-c", str(metal_src), "-o", str(BUILD_DIR / "batch_cgemm.air")],
|
|
89
|
+
check=True, capture_output=True,
|
|
90
|
+
)
|
|
91
|
+
subprocess.run(
|
|
92
|
+
["xcrun", "-sdk", "macosx", "metallib", str(BUILD_DIR / "batch_cgemm.air"), "-o", str(metallib_path)],
|
|
93
|
+
check=True, capture_output=True,
|
|
94
|
+
)
|
|
95
|
+
except Exception as exc:
|
|
96
|
+
warnings.warn(f"Metal shader precompilation skipped: {exc}", RuntimeWarning, stacklevel=2)
|
|
97
|
+
|
|
98
|
+
stale = not lib_path.exists() or any(lib_path.stat().st_mtime < s.stat().st_mtime for s in sources if s.exists())
|
|
99
|
+
if stale:
|
|
100
|
+
frameworks = ["-framework", "Accelerate", "-framework", "Metal", "-framework", "Foundation"] if is_macos() else ["-lblas"]
|
|
101
|
+
cmd = [
|
|
102
|
+
compiler, "-std=c11", "-O3", "-fPIC", "-shared",
|
|
103
|
+
"-D_DEFAULT_SOURCE", "-Wno-unused-function", "-Wno-unused-variable",
|
|
104
|
+
f"-I{NATIVE_DIR}",
|
|
105
|
+
*[str(s) for s in sources if s.exists()],
|
|
106
|
+
*frameworks, "-lm", "-o", str(lib_path),
|
|
107
|
+
]
|
|
108
|
+
res = subprocess.run(cmd, capture_output=True, text=True)
|
|
109
|
+
if res.returncode != 0:
|
|
110
|
+
warnings.warn(
|
|
111
|
+
f"Failed to build faster-diffbloch native library with {compiler}:\n{res.stderr}\n"
|
|
112
|
+
"Falling back to standard PyTorch execution.",
|
|
113
|
+
RuntimeWarning,
|
|
114
|
+
stacklevel=2,
|
|
115
|
+
)
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
try:
|
|
119
|
+
return ctypes.CDLL(str(lib_path))
|
|
120
|
+
except Exception as exc:
|
|
121
|
+
warnings.warn(f"Could not load faster-diffbloch library: {exc}", RuntimeWarning, stacklevel=2)
|
|
122
|
+
return None
|
faster_diffbloch-0.1.0/PKG-INFO
DELETED
|
@@ -1,98 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: faster-diffbloch
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: Drop-in Metal GPU and CPU acceleration for diffBloch electron crystallography
|
|
5
|
-
Project-URL: Homepage, https://godofecht.github.io/diffFlow/
|
|
6
|
-
Project-URL: Repository, https://github.com/godofecht/diffFlow
|
|
7
|
-
Project-URL: Issues, https://github.com/godofecht/diffFlow/issues
|
|
8
|
-
Project-URL: Original diffBloch, https://diffbloch.com
|
|
9
|
-
Author-email: Abhishek Shivakumar <abhishek@example.com>
|
|
10
|
-
License-Expression: MIT
|
|
11
|
-
License-File: LICENSE
|
|
12
|
-
Classifier: Development Status :: 4 - Beta
|
|
13
|
-
Classifier: Intended Audience :: Science/Research
|
|
14
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
-
Classifier: Programming Language :: Python :: 3
|
|
16
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
-
Classifier: Topic :: Scientific/Engineering :: Physics
|
|
20
|
-
Requires-Python: >=3.10
|
|
21
|
-
Requires-Dist: numpy>=1.24
|
|
22
|
-
Requires-Dist: torch>=2.0
|
|
23
|
-
Provides-Extra: diffbloch
|
|
24
|
-
Requires-Dist: diffbloch; extra == 'diffbloch'
|
|
25
|
-
Description-Content-Type: text/markdown
|
|
26
|
-
|
|
27
|
-
# faster-diffBloch
|
|
28
|
-
|
|
29
|
-
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
30
|
-
|
|
31
|
-
Documentation and comparison benchmarks: [https://godofecht.github.io/diffFlow/](https://godofecht.github.io/diffFlow/)
|
|
32
|
-
|
|
33
|
-
Original diffBloch project: [https://diffbloch.com](https://diffbloch.com)
|
|
34
|
-
|
|
35
|
-
---
|
|
36
|
-
|
|
37
|
-
## Why faster-diffBloch?
|
|
38
|
-
|
|
39
|
-
1. **Native Metal GPU Execution:**
|
|
40
|
-
PyTorch MPS lacks a native GPU kernel for `aten::linalg_matrix_exp`, which causes PyTorch to fall back to CPU execution with host-device memory transfers. `faster-diffBloch` executes matrix exponentials directly on Apple Silicon Metal with zero-copy unified memory.
|
|
41
|
-
|
|
42
|
-
2. **Blocked-Pair Adjoint Formulation:**
|
|
43
|
-
Standard matrix exponential autograd embeds the operator into a $2N \times 2N$ block matrix, costing $8 \times N^3$ FLOPs. `faster-diffBloch` evaluates the pullback in the block-triangular pair algebra $(Y_a Y_b, Y_a L_b + L_a Y_b)$, reducing the work to $3 \times N^3$ FLOPs (2.67x fewer products).
|
|
44
|
-
|
|
45
|
-
3. **Bit-for-Bit Validation:**
|
|
46
|
-
Passes all 738 unit tests in diffBloch and reproduces the experimental 99-rotation quartz dataset ($R_{\text{obs}} = 0.0486$).
|
|
47
|
-
|
|
48
|
-
---
|
|
49
|
-
|
|
50
|
-
## Performance
|
|
51
|
-
|
|
52
|
-
Forward and backward timing comparison on Apple Silicon (M4 Max) at $N=579$ beams (CsPbBr3 scale):
|
|
53
|
-
|
|
54
|
-
| Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
|
|
55
|
-
| :--- | :---: | :---: | :---: | :---: |
|
|
56
|
-
| PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
|
|
57
|
-
| PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
|
|
58
|
-
| **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
|
|
59
|
-
| **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
|
|
60
|
-
|
|
61
|
-
---
|
|
62
|
-
|
|
63
|
-
## Installation
|
|
64
|
-
|
|
65
|
-
```bash
|
|
66
|
-
pip install faster-diffbloch
|
|
67
|
-
```
|
|
68
|
-
|
|
69
|
-
---
|
|
70
|
-
|
|
71
|
-
## Usage
|
|
72
|
-
|
|
73
|
-
### 1. Drop-in CLI
|
|
74
|
-
|
|
75
|
-
Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
|
|
76
|
-
|
|
77
|
-
```bash
|
|
78
|
-
diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
79
|
-
diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
80
|
-
```
|
|
81
|
-
|
|
82
|
-
### 2. Python API Injection
|
|
83
|
-
|
|
84
|
-
Enable acceleration inside any existing diffBloch script:
|
|
85
|
-
|
|
86
|
-
```python
|
|
87
|
-
import faster_diffbloch
|
|
88
|
-
|
|
89
|
-
# Enable Metal GPU acceleration
|
|
90
|
-
faster_diffbloch.enable(device="gpu")
|
|
91
|
-
|
|
92
|
-
# Or CPU acceleration
|
|
93
|
-
faster_diffbloch.enable(device="cpu")
|
|
94
|
-
|
|
95
|
-
# Run standard diffBloch code
|
|
96
|
-
import diffBloch
|
|
97
|
-
# All propagate and matrix_exp calls now route through faster-diffBloch
|
|
98
|
-
```
|
faster_diffbloch-0.1.0/README.md
DELETED
|
@@ -1,72 +0,0 @@
|
|
|
1
|
-
# faster-diffBloch
|
|
2
|
-
|
|
3
|
-
Drop-in Apple Silicon Metal GPU and optimized CPU acceleration for [diffBloch](https://diffbloch.com) electron crystallography structure refinement.
|
|
4
|
-
|
|
5
|
-
Documentation and comparison benchmarks: [https://godofecht.github.io/diffFlow/](https://godofecht.github.io/diffFlow/)
|
|
6
|
-
|
|
7
|
-
Original diffBloch project: [https://diffbloch.com](https://diffbloch.com)
|
|
8
|
-
|
|
9
|
-
---
|
|
10
|
-
|
|
11
|
-
## Why faster-diffBloch?
|
|
12
|
-
|
|
13
|
-
1. **Native Metal GPU Execution:**
|
|
14
|
-
PyTorch MPS lacks a native GPU kernel for `aten::linalg_matrix_exp`, which causes PyTorch to fall back to CPU execution with host-device memory transfers. `faster-diffBloch` executes matrix exponentials directly on Apple Silicon Metal with zero-copy unified memory.
|
|
15
|
-
|
|
16
|
-
2. **Blocked-Pair Adjoint Formulation:**
|
|
17
|
-
Standard matrix exponential autograd embeds the operator into a $2N \times 2N$ block matrix, costing $8 \times N^3$ FLOPs. `faster-diffBloch` evaluates the pullback in the block-triangular pair algebra $(Y_a Y_b, Y_a L_b + L_a Y_b)$, reducing the work to $3 \times N^3$ FLOPs (2.67x fewer products).
|
|
18
|
-
|
|
19
|
-
3. **Bit-for-Bit Validation:**
|
|
20
|
-
Passes all 738 unit tests in diffBloch and reproduces the experimental 99-rotation quartz dataset ($R_{\text{obs}} = 0.0486$).
|
|
21
|
-
|
|
22
|
-
---
|
|
23
|
-
|
|
24
|
-
## Performance
|
|
25
|
-
|
|
26
|
-
Forward and backward timing comparison on Apple Silicon (M4 Max) at $N=579$ beams (CsPbBr3 scale):
|
|
27
|
-
|
|
28
|
-
| Implementation | Forward | Forward + Backward | Speedup vs PyTorch CPU | Speedup vs PyTorch MPS |
|
|
29
|
-
| :--- | :---: | :---: | :---: | :---: |
|
|
30
|
-
| PyTorch CPU | 25.7 ms | 130.3 ms | 1.00x | 1.17x |
|
|
31
|
-
| PyTorch MPS (fallback) | 26.2 ms | 153.0 ms | 0.85x | 1.00x |
|
|
32
|
-
| **faster-diffBloch CPU** | **24.4 ms** | **83.5 ms** | **1.56x** | **1.83x** |
|
|
33
|
-
| **faster-diffBloch Metal GPU** | **13.1 ms** | **58.1 ms** | **2.24x** | **2.63x** |
|
|
34
|
-
|
|
35
|
-
---
|
|
36
|
-
|
|
37
|
-
## Installation
|
|
38
|
-
|
|
39
|
-
```bash
|
|
40
|
-
pip install faster-diffbloch
|
|
41
|
-
```
|
|
42
|
-
|
|
43
|
-
---
|
|
44
|
-
|
|
45
|
-
## Usage
|
|
46
|
-
|
|
47
|
-
### 1. Drop-in CLI
|
|
48
|
-
|
|
49
|
-
Use `diffbloch-fast` or `faster-diffbloch` anywhere you would use `diffbloch`:
|
|
50
|
-
|
|
51
|
-
```bash
|
|
52
|
-
diffbloch-fast infer examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
53
|
-
diffbloch-fast refine examples/Colmey_et_al_2026/data/quartz-no-abs
|
|
54
|
-
```
|
|
55
|
-
|
|
56
|
-
### 2. Python API Injection
|
|
57
|
-
|
|
58
|
-
Enable acceleration inside any existing diffBloch script:
|
|
59
|
-
|
|
60
|
-
```python
|
|
61
|
-
import faster_diffbloch
|
|
62
|
-
|
|
63
|
-
# Enable Metal GPU acceleration
|
|
64
|
-
faster_diffbloch.enable(device="gpu")
|
|
65
|
-
|
|
66
|
-
# Or CPU acceleration
|
|
67
|
-
faster_diffbloch.enable(device="cpu")
|
|
68
|
-
|
|
69
|
-
# Run standard diffBloch code
|
|
70
|
-
import diffBloch
|
|
71
|
-
# All propagate and matrix_exp calls now route through faster-diffBloch
|
|
72
|
-
```
|
|
@@ -1,58 +0,0 @@
|
|
|
1
|
-
"""Native runtime builder and loader for faster-diffbloch."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
import ctypes
|
|
6
|
-
import os
|
|
7
|
-
import subprocess
|
|
8
|
-
import sys
|
|
9
|
-
from pathlib import Path
|
|
10
|
-
|
|
11
|
-
NATIVE_DIR = Path(__file__).resolve().parent / "native"
|
|
12
|
-
BUILD_DIR = Path.home() / ".cache" / "faster_diffbloch"
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
def build_and_load_library() -> ctypes.CDLL:
|
|
16
|
-
"""Build or load the cached native acceleration library."""
|
|
17
|
-
BUILD_DIR.mkdir(parents=True, exist_ok=True)
|
|
18
|
-
lib_name = "libfaster_diffbloch.dylib" if sys.platform == "darwin" else "libfaster_diffbloch.so"
|
|
19
|
-
lib_path = BUILD_DIR / lib_name
|
|
20
|
-
|
|
21
|
-
sources = [
|
|
22
|
-
NATIVE_DIR / "bridge_lib.c",
|
|
23
|
-
NATIVE_DIR / "batch_cgemm.c",
|
|
24
|
-
NATIVE_DIR / "native_scattering.c",
|
|
25
|
-
]
|
|
26
|
-
if sys.platform == "darwin":
|
|
27
|
-
sources.append(NATIVE_DIR / "metal_batch_cgemm.m")
|
|
28
|
-
# Build metallib if metal is available
|
|
29
|
-
metal_src = NATIVE_DIR / "batch_cgemm.metal"
|
|
30
|
-
metallib_path = BUILD_DIR / "batch_cgemm.metallib"
|
|
31
|
-
if metal_src.exists() and (not metallib_path.exists() or metallib_path.stat().st_mtime < metal_src.stat().st_mtime):
|
|
32
|
-
try:
|
|
33
|
-
subprocess.run(
|
|
34
|
-
["xcrun", "-sdk", "macosx", "metal", "-O3", "-c", str(metal_src), "-o", str(BUILD_DIR / "batch_cgemm.air")],
|
|
35
|
-
check=True, capture_output=True
|
|
36
|
-
)
|
|
37
|
-
subprocess.run(
|
|
38
|
-
["xcrun", "-sdk", "macosx", "metallib", str(BUILD_DIR / "batch_cgemm.air"), "-o", str(metallib_path)],
|
|
39
|
-
check=True, capture_output=True
|
|
40
|
-
)
|
|
41
|
-
except Exception:
|
|
42
|
-
pass
|
|
43
|
-
|
|
44
|
-
stale = not lib_path.exists() or any(lib_path.stat().st_mtime < s.stat().st_mtime for s in sources if s.exists())
|
|
45
|
-
if stale:
|
|
46
|
-
frameworks = ["-framework", "Accelerate", "-framework", "Metal", "-framework", "Foundation"] if sys.platform == "darwin" else ["-lblas"]
|
|
47
|
-
cmd = [
|
|
48
|
-
"clang", "-std=c11", "-O3", "-fPIC", "-shared",
|
|
49
|
-
"-D_DEFAULT_SOURCE", "-Wno-unused-function", "-Wno-unused-variable",
|
|
50
|
-
f"-I{NATIVE_DIR}",
|
|
51
|
-
*[str(s) for s in sources if s.exists()],
|
|
52
|
-
*frameworks, "-lm", "-o", str(lib_path),
|
|
53
|
-
]
|
|
54
|
-
res = subprocess.run(cmd, capture_output=True, text=True)
|
|
55
|
-
if res.returncode != 0:
|
|
56
|
-
raise RuntimeError(f"Failed to build faster-diffbloch native library:\n{res.stderr}")
|
|
57
|
-
|
|
58
|
-
return ctypes.CDLL(str(lib_path))
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/batch_cgemm.metal
RENAMED
|
File without changes
|
|
File without changes
|
{faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/metal_batch_cgemm.m
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.c
RENAMED
|
File without changes
|
{faster_diffbloch-0.1.0 → faster_diffbloch-0.1.2}/src/faster_diffbloch/native/native_scattering.h
RENAMED
|
File without changes
|