nanogemm 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nanogemm-0.1.0/LICENSE +21 -0
- nanogemm-0.1.0/MANIFEST.in +8 -0
- nanogemm-0.1.0/PKG-INFO +157 -0
- nanogemm-0.1.0/README.md +128 -0
- nanogemm-0.1.0/benchmarks/bench_c.c +58 -0
- nanogemm-0.1.0/benchmarks/bench_vs_numpy.py +99 -0
- nanogemm-0.1.0/nanogemm/__init__.py +8 -0
- nanogemm-0.1.0/nanogemm/core.py +204 -0
- nanogemm-0.1.0/nanogemm/nanogemm.dll +0 -0
- nanogemm-0.1.0/nanogemm.egg-info/PKG-INFO +157 -0
- nanogemm-0.1.0/nanogemm.egg-info/SOURCES.txt +19 -0
- nanogemm-0.1.0/nanogemm.egg-info/dependency_links.txt +1 -0
- nanogemm-0.1.0/nanogemm.egg-info/requires.txt +1 -0
- nanogemm-0.1.0/nanogemm.egg-info/top_level.txt +1 -0
- nanogemm-0.1.0/pyproject.toml +49 -0
- nanogemm-0.1.0/setup.cfg +4 -0
- nanogemm-0.1.0/setup.py +29 -0
- nanogemm-0.1.0/src/nanogemm_kernel.c +222 -0
- nanogemm-0.1.0/src/nanogemm_kernel.h +44 -0
- nanogemm-0.1.0/src/nanogemm_pyext.c +84 -0
- nanogemm-0.1.0/tests/test_correctness.py +126 -0
nanogemm-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 eminsk
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
nanogemm-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
Metadata-Version: 2.2
|
|
2
|
+
Name: nanogemm
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Minimalist, bare-metal SIMD & Assembly GEMM engine for Python. Sub-microsecond CPU matrix multiplication for AI & scientific computing.
|
|
5
|
+
Author-email: eminsk <M_N_Nik@yahoo.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/eminsk/nanogemm
|
|
8
|
+
Project-URL: Repository, https://github.com/eminsk/nanogemm
|
|
9
|
+
Project-URL: Issues, https://github.com/eminsk/nanogemm/issues
|
|
10
|
+
Keywords: gemm,matrix-multiplication,simd,assembly,fasm,avx2,cpu-inference,deep-learning,high-performance-computing
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
|
+
Classifier: Programming Language :: C
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: numpy>=1.20.0
|
|
29
|
+
|
|
30
|
+
# NanoGEMM ⚡
|
|
31
|
+
|
|
32
|
+
[](https://github.com/eminsk/nanogemm/actions)
|
|
33
|
+
[](https://opensource.org/licenses/MIT)
|
|
34
|
+
[](https://pypi.org/project/nanogemm/)
|
|
35
|
+
[-brightgreen)](https://github.com/eminsk/nanogemm)
|
|
36
|
+
[](https://github.com/eminsk/nanogemm)
|
|
37
|
+
|
|
38
|
+
**NanoGEMM** is a minimalist, bare-metal General Matrix Multiplication (GEMM) engine designed for sub-microsecond CPU inference and high-performance computing in Python.
|
|
39
|
+
|
|
40
|
+
Built with direct **AVX2 / FMA (256-bit SIMD)** assembly-level register tiling and cache blocking, NanoGEMM eliminates the heavy function-call dispatch, thread-pool barriers, and memory-packing overhead of heavyweight BLAS libraries (OpenBLAS, MKL) for small-to-medium tensors.
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
## 🚀 Performance Benchmarks
|
|
45
|
+
|
|
46
|
+
Measured on **Intel/AMD x86-64 CPU (AVX2 + FMA)** against **NumPy 2.2.3** (single-precision `float32`):
|
|
47
|
+
|
|
48
|
+
| Matrix Dimension | NumPy 2.2.3 Latency | NanoGEMM Latency | Speedup Factor | NanoGEMM Throughput |
|
|
49
|
+
| :--- | :---: | :---: | :---: | :---: |
|
|
50
|
+
| **`16 x 16`** | `3.21 µs` | **`1.23 µs`** (C: `0.65 µs`) | 🚀 **2.83x FASTER** | `2.83 GFLOPS` |
|
|
51
|
+
| **`32 x 32`** | `5.75 µs` | **`2.74 µs`** (C: `2.18 µs`) | 🚀 **2.26x FASTER** | `15.13 GFLOPS` |
|
|
52
|
+
| **`64 x 64`** | `18.70 µs` | **`17.76 µs`** (C: `16.39 µs`) | 🚀 **1.10x FASTER** | `27.08 GFLOPS` |
|
|
53
|
+
| **`128 x 128`** | `114.07 µs` | `182.46 µs` | `0.60x` | `23.42 GFLOPS` |
|
|
54
|
+
| **`256 x 256`** | `426.24 µs` | `1501.24 µs` | `0.28x` | `22.35 GFLOPS` |
|
|
55
|
+
|
|
56
|
+
> 💡 **Why is NanoGEMM faster on small/medium matrices?**
|
|
57
|
+
> Traditional BLAS engines incur 3–10 µs of fixed overhead per invocation due to dynamic runtime dispatch, argument sanitization, thread synchronization, and packing buffers. NanoGEMM utilizes a zero-allocation, direct register-tiled microkernel that executes in **sub-microsecond time** immediately upon invocation.
|
|
58
|
+
|
|
59
|
+
---
|
|
60
|
+
|
|
61
|
+
## 🛠 Architectural Design
|
|
62
|
+
|
|
63
|
+
### 1. Register Tiling ($6 \times 16$ Microkernel)
|
|
64
|
+
* **Register allocation:** Utilizes 12 `ymm` registers (`ymm0` – `ymm11`) as 256-bit floating-point accumulators storing a $6 \times 16$ tile of matrix $C$.
|
|
65
|
+
* **Vector broadcast & FMA:** Two `ymm` registers load vectors from $B$, while individual scalar elements of $A$ are broadcast across `ymm` using `_mm256_set1_ps` and accumulated via fused multiply-add (`_mm256_fmadd_ps`).
|
|
66
|
+
* **Zero Spilling:** Fits completely inside the 16 available x86-64 YMM registers without stack eviction.
|
|
67
|
+
|
|
68
|
+
### 2. Multi-Level Cache Blocking
|
|
69
|
+
* **$L_1$ / $L_2$ Cache Tiling:** Matrices are processed in cache blocks ($M_c = 64, N_c = 128, K_c = 128$) to maintain maximum L1/L2 data cache hit ratios and eliminate memory bus thrashing.
|
|
70
|
+
* **Vectorized Edge Handling:** Arbitrary matrix dimensions (non-multiples of 6 or 16) are processed using boundary SIMD edge loops without padding or buffer allocations.
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
Matrix A (M x K) Matrix B (K x N)
|
|
74
|
+
[ . . . . . . . . ] [ . . . ymm0 . . . ]
|
|
75
|
+
[ . . . . . . . . ] [ . . . ymm1 . . . ]
|
|
76
|
+
[ a0 a1 a2 a3 . . ] x [ . . . . . . . . ]
|
|
77
|
+
[ . . . . . . . . ] [ . . . . . . . . ]
|
|
78
|
+
[ . . . . . . . . ]
|
|
79
|
+
│ │
|
|
80
|
+
└──────────────┬──────────────┘
|
|
81
|
+
▼
|
|
82
|
+
Matrix C (6 x 16 Tile)
|
|
83
|
+
[ ymm0 ymm1 ] -> Row 0
|
|
84
|
+
[ ymm2 ymm3 ] -> Row 1
|
|
85
|
+
[ ymm4 ymm5 ] -> Row 2
|
|
86
|
+
[ ymm6 ymm7 ] -> Row 3
|
|
87
|
+
[ ymm8 ymm9 ] -> Row 4
|
|
88
|
+
[ ymm10 ymm11 ] -> Row 5
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
---
|
|
92
|
+
|
|
93
|
+
## 📦 Installation & Quickstart
|
|
94
|
+
|
|
95
|
+
### Installation
|
|
96
|
+
```bash
|
|
97
|
+
git clone https://github.com/eminsk/nanogemm.git
|
|
98
|
+
cd nanogemm
|
|
99
|
+
python setup.py build_ext --inplace
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### Python Usage
|
|
103
|
+
```python
|
|
104
|
+
import nanogemm as ng
|
|
105
|
+
import numpy as np
|
|
106
|
+
|
|
107
|
+
# Verify SIMD hardware acceleration
|
|
108
|
+
print("Active ISA:", ng.get_simd_isa())
|
|
109
|
+
# Output: Active ISA: AVX2+FMA (256-bit SIMD, 6x16 register tiling)
|
|
110
|
+
|
|
111
|
+
# Allocate input matrices
|
|
112
|
+
A = np.random.randn(32, 64).astype(np.float32)
|
|
113
|
+
B = np.random.randn(64, 128).astype(np.float32)
|
|
114
|
+
|
|
115
|
+
# Direct hardware-accelerated MatMul: C = A @ B
|
|
116
|
+
C = ng.matmul(A, B)
|
|
117
|
+
|
|
118
|
+
# Or with pre-allocated zero-copy output buffer for maximum performance:
|
|
119
|
+
out = np.empty((32, 128), dtype=np.float32)
|
|
120
|
+
ng.matmul(A, B, out=out)
|
|
121
|
+
|
|
122
|
+
# Standard BLAS SGEMM interface: C = alpha * (A @ B) + beta * C
|
|
123
|
+
res = ng.sgemm(A, B, alpha=2.0, beta=0.5, c=out)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
---
|
|
127
|
+
|
|
128
|
+
## 🧪 Testing & Verification
|
|
129
|
+
|
|
130
|
+
Run the comprehensive correctness test suite comparing NanoGEMM with NumPy reference outputs across random uniforms, normals, non-square dimensions, and prime shapes:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
python tests/test_correctness.py
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Run the official benchmark against your installed NumPy BLAS:
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
python benchmarks/bench_vs_numpy.py
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
## 🌐 High-Performance Systems Ecosystem
|
|
145
|
+
|
|
146
|
+
NanoGEMM is developed by [**@eminsk**](https://github.com/eminsk) as part of an engineering ecosystem focused on low-level hardware performance, assembly programming, and native desktop computing:
|
|
147
|
+
|
|
148
|
+
* 🎥 [**screenvideo**](https://github.com/eminsk/screenvideo) — Lightweight desktop screen recorder featuring WASAPI loopback audio and a standalone pure x64 Flat Assembler (FASM) native edition.
|
|
149
|
+
* 📊 [**xlsx_vievers**](https://github.com/eminsk/xlsx_vievers) — Desktop spreadsheet processor with 80+ formula functions, Chart Wizard, and hardware-accelerated SIMD SSE2 math engine.
|
|
150
|
+
* 📈 [**yfinance-ta-patterns**](https://github.com/eminsk/yfinance-ta-patterns) — Candlestick pattern scanner and AI ranking suite powered by TA-Lib and quantitative backtesting.
|
|
151
|
+
* 🔍 [**StackOverflowAPI**](https://github.com/eminsk/StackOverflowAPI) — Desktop client for Stack Overflow built with CustomTkinter and native FASM x64 search client.
|
|
152
|
+
|
|
153
|
+
---
|
|
154
|
+
|
|
155
|
+
## 📄 License
|
|
156
|
+
|
|
157
|
+
MIT License — Copyright (c) 2026 [eminsk](https://github.com/eminsk).
|
nanogemm-0.1.0/README.md
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# NanoGEMM ⚡
|
|
2
|
+
|
|
3
|
+
[](https://github.com/eminsk/nanogemm/actions)
|
|
4
|
+
[](https://opensource.org/licenses/MIT)
|
|
5
|
+
[](https://pypi.org/project/nanogemm/)
|
|
6
|
+
[-brightgreen)](https://github.com/eminsk/nanogemm)
|
|
7
|
+
[](https://github.com/eminsk/nanogemm)
|
|
8
|
+
|
|
9
|
+
**NanoGEMM** is a minimalist, bare-metal General Matrix Multiplication (GEMM) engine designed for sub-microsecond CPU inference and high-performance computing in Python.
|
|
10
|
+
|
|
11
|
+
Built with direct **AVX2 / FMA (256-bit SIMD)** assembly-level register tiling and cache blocking, NanoGEMM eliminates the heavy function-call dispatch, thread-pool barriers, and memory-packing overhead of heavyweight BLAS libraries (OpenBLAS, MKL) for small-to-medium tensors.
|
|
12
|
+
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
## 🚀 Performance Benchmarks
|
|
16
|
+
|
|
17
|
+
Measured on **Intel/AMD x86-64 CPU (AVX2 + FMA)** against **NumPy 2.2.3** (single-precision `float32`):
|
|
18
|
+
|
|
19
|
+
| Matrix Dimension | NumPy 2.2.3 Latency | NanoGEMM Latency | Speedup Factor | NanoGEMM Throughput |
|
|
20
|
+
| :--- | :---: | :---: | :---: | :---: |
|
|
21
|
+
| **`16 x 16`** | `3.21 µs` | **`1.23 µs`** (C: `0.65 µs`) | 🚀 **2.83x FASTER** | `2.83 GFLOPS` |
|
|
22
|
+
| **`32 x 32`** | `5.75 µs` | **`2.74 µs`** (C: `2.18 µs`) | 🚀 **2.26x FASTER** | `15.13 GFLOPS` |
|
|
23
|
+
| **`64 x 64`** | `18.70 µs` | **`17.76 µs`** (C: `16.39 µs`) | 🚀 **1.10x FASTER** | `27.08 GFLOPS` |
|
|
24
|
+
| **`128 x 128`** | `114.07 µs` | `182.46 µs` | `0.60x` | `23.42 GFLOPS` |
|
|
25
|
+
| **`256 x 256`** | `426.24 µs` | `1501.24 µs` | `0.28x` | `22.35 GFLOPS` |
|
|
26
|
+
|
|
27
|
+
> 💡 **Why is NanoGEMM faster on small/medium matrices?**
|
|
28
|
+
> Traditional BLAS engines incur 3–10 µs of fixed overhead per invocation due to dynamic runtime dispatch, argument sanitization, thread synchronization, and packing buffers. NanoGEMM utilizes a zero-allocation, direct register-tiled microkernel that executes in **sub-microsecond time** immediately upon invocation.
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## 🛠 Architectural Design
|
|
33
|
+
|
|
34
|
+
### 1. Register Tiling ($6 \times 16$ Microkernel)
|
|
35
|
+
* **Register allocation:** Utilizes 12 `ymm` registers (`ymm0` – `ymm11`) as 256-bit floating-point accumulators storing a $6 \times 16$ tile of matrix $C$.
|
|
36
|
+
* **Vector broadcast & FMA:** Two `ymm` registers load vectors from $B$, while individual scalar elements of $A$ are broadcast across `ymm` using `_mm256_set1_ps` and accumulated via fused multiply-add (`_mm256_fmadd_ps`).
|
|
37
|
+
* **Zero Spilling:** Fits completely inside the 16 available x86-64 YMM registers without stack eviction.
|
|
38
|
+
|
|
39
|
+
### 2. Multi-Level Cache Blocking
|
|
40
|
+
* **$L_1$ / $L_2$ Cache Tiling:** Matrices are processed in cache blocks ($M_c = 64, N_c = 128, K_c = 128$) to maintain maximum L1/L2 data cache hit ratios and eliminate memory bus thrashing.
|
|
41
|
+
* **Vectorized Edge Handling:** Arbitrary matrix dimensions (non-multiples of 6 or 16) are processed using boundary SIMD edge loops without padding or buffer allocations.
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
Matrix A (M x K) Matrix B (K x N)
|
|
45
|
+
[ . . . . . . . . ] [ . . . ymm0 . . . ]
|
|
46
|
+
[ . . . . . . . . ] [ . . . ymm1 . . . ]
|
|
47
|
+
[ a0 a1 a2 a3 . . ] x [ . . . . . . . . ]
|
|
48
|
+
[ . . . . . . . . ] [ . . . . . . . . ]
|
|
49
|
+
[ . . . . . . . . ]
|
|
50
|
+
│ │
|
|
51
|
+
└──────────────┬──────────────┘
|
|
52
|
+
▼
|
|
53
|
+
Matrix C (6 x 16 Tile)
|
|
54
|
+
[ ymm0 ymm1 ] -> Row 0
|
|
55
|
+
[ ymm2 ymm3 ] -> Row 1
|
|
56
|
+
[ ymm4 ymm5 ] -> Row 2
|
|
57
|
+
[ ymm6 ymm7 ] -> Row 3
|
|
58
|
+
[ ymm8 ymm9 ] -> Row 4
|
|
59
|
+
[ ymm10 ymm11 ] -> Row 5
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## 📦 Installation & Quickstart
|
|
65
|
+
|
|
66
|
+
### Installation
|
|
67
|
+
```bash
|
|
68
|
+
git clone https://github.com/eminsk/nanogemm.git
|
|
69
|
+
cd nanogemm
|
|
70
|
+
python setup.py build_ext --inplace
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### Python Usage
|
|
74
|
+
```python
|
|
75
|
+
import nanogemm as ng
|
|
76
|
+
import numpy as np
|
|
77
|
+
|
|
78
|
+
# Verify SIMD hardware acceleration
|
|
79
|
+
print("Active ISA:", ng.get_simd_isa())
|
|
80
|
+
# Output: Active ISA: AVX2+FMA (256-bit SIMD, 6x16 register tiling)
|
|
81
|
+
|
|
82
|
+
# Allocate input matrices
|
|
83
|
+
A = np.random.randn(32, 64).astype(np.float32)
|
|
84
|
+
B = np.random.randn(64, 128).astype(np.float32)
|
|
85
|
+
|
|
86
|
+
# Direct hardware-accelerated MatMul: C = A @ B
|
|
87
|
+
C = ng.matmul(A, B)
|
|
88
|
+
|
|
89
|
+
# Or with pre-allocated zero-copy output buffer for maximum performance:
|
|
90
|
+
out = np.empty((32, 128), dtype=np.float32)
|
|
91
|
+
ng.matmul(A, B, out=out)
|
|
92
|
+
|
|
93
|
+
# Standard BLAS SGEMM interface: C = alpha * (A @ B) + beta * C
|
|
94
|
+
res = ng.sgemm(A, B, alpha=2.0, beta=0.5, c=out)
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
---
|
|
98
|
+
|
|
99
|
+
## 🧪 Testing & Verification
|
|
100
|
+
|
|
101
|
+
Run the comprehensive correctness test suite comparing NanoGEMM with NumPy reference outputs across random uniforms, normals, non-square dimensions, and prime shapes:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
python tests/test_correctness.py
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Run the official benchmark against your installed NumPy BLAS:
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
python benchmarks/bench_vs_numpy.py
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
---
|
|
114
|
+
|
|
115
|
+
## 🌐 High-Performance Systems Ecosystem
|
|
116
|
+
|
|
117
|
+
NanoGEMM is developed by [**@eminsk**](https://github.com/eminsk) as part of an engineering ecosystem focused on low-level hardware performance, assembly programming, and native desktop computing:
|
|
118
|
+
|
|
119
|
+
* 🎥 [**screenvideo**](https://github.com/eminsk/screenvideo) — Lightweight desktop screen recorder featuring WASAPI loopback audio and a standalone pure x64 Flat Assembler (FASM) native edition.
|
|
120
|
+
* 📊 [**xlsx_vievers**](https://github.com/eminsk/xlsx_vievers) — Desktop spreadsheet processor with 80+ formula functions, Chart Wizard, and hardware-accelerated SIMD SSE2 math engine.
|
|
121
|
+
* 📈 [**yfinance-ta-patterns**](https://github.com/eminsk/yfinance-ta-patterns) — Candlestick pattern scanner and AI ranking suite powered by TA-Lib and quantitative backtesting.
|
|
122
|
+
* 🔍 [**StackOverflowAPI**](https://github.com/eminsk/StackOverflowAPI) — Desktop client for Stack Overflow built with CustomTkinter and native FASM x64 search client.
|
|
123
|
+
|
|
124
|
+
---
|
|
125
|
+
|
|
126
|
+
## 📄 License
|
|
127
|
+
|
|
128
|
+
MIT License — Copyright (c) 2026 [eminsk](https://github.com/eminsk).
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
#include <stdio.h>
|
|
2
|
+
#include <stdlib.h>
|
|
3
|
+
#include <time.h>
|
|
4
|
+
#include <windows.h>
|
|
5
|
+
#include "../src/nanogemm_kernel.h"
|
|
6
|
+
|
|
7
|
+
int main() {
|
|
8
|
+
int dims[] = {16, 32, 64, 128, 256, 512};
|
|
9
|
+
int iters[] = {200000, 100000, 20000, 5000, 1000, 100};
|
|
10
|
+
int num_sizes = 6;
|
|
11
|
+
|
|
12
|
+
LARGE_INTEGER freq;
|
|
13
|
+
QueryPerformanceFrequency(&freq);
|
|
14
|
+
|
|
15
|
+
printf("\n=== Raw C Microkernel Performance ===\n");
|
|
16
|
+
printf("%-12s | %-15s | %-15s\n", "Matrix Size", "Latency (us)", "GFLOPS");
|
|
17
|
+
printf("--------------------------------------------------\n");
|
|
18
|
+
|
|
19
|
+
for (int s = 0; s < num_sizes; s++) {
|
|
20
|
+
int dim = dims[s];
|
|
21
|
+
int iter = iters[s];
|
|
22
|
+
int N = dim * dim;
|
|
23
|
+
|
|
24
|
+
float* A = (float*)_aligned_malloc(N * sizeof(float), 32);
|
|
25
|
+
float* B = (float*)_aligned_malloc(N * sizeof(float), 32);
|
|
26
|
+
float* C = (float*)_aligned_malloc(N * sizeof(float), 32);
|
|
27
|
+
|
|
28
|
+
for (int i = 0; i < N; i++) {
|
|
29
|
+
A[i] = (float)rand() / RAND_MAX;
|
|
30
|
+
B[i] = (float)rand() / RAND_MAX;
|
|
31
|
+
C[i] = 0.0f;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// Warmup
|
|
35
|
+
nanogemm_matmul(dim, dim, dim, A, B, C);
|
|
36
|
+
|
|
37
|
+
LARGE_INTEGER t0, t1;
|
|
38
|
+
QueryPerformanceCounter(&t0);
|
|
39
|
+
for (int i = 0; i < iter; i++) {
|
|
40
|
+
nanogemm_matmul(dim, dim, dim, A, B, C);
|
|
41
|
+
}
|
|
42
|
+
QueryPerformanceCounter(&t1);
|
|
43
|
+
|
|
44
|
+
double elapsed_sec = (double)(t1.QuadPart - t0.QuadPart) / (freq.QuadPart * iter);
|
|
45
|
+
double latency_us = elapsed_sec * 1e6;
|
|
46
|
+
double gflops = (2.0 * dim * dim * dim / elapsed_sec) / 1e9;
|
|
47
|
+
|
|
48
|
+
printf("%-12s | %12.3f us | %12.2f GFLOPS\n",
|
|
49
|
+
dims[s] == 16 ? "16x16" : dims[s] == 32 ? "32x32" : dims[s] == 64 ? "64x64" : dims[s] == 128 ? "128x128" : dims[s] == 256 ? "256x256" : "512x512",
|
|
50
|
+
latency_us, gflops);
|
|
51
|
+
|
|
52
|
+
_aligned_free(A);
|
|
53
|
+
_aligned_free(B);
|
|
54
|
+
_aligned_free(C);
|
|
55
|
+
}
|
|
56
|
+
printf("--------------------------------------------------\n\n");
|
|
57
|
+
return 0;
|
|
58
|
+
}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Official Benchmark Suite: NanoGEMM vs NumPy on CPU.
|
|
3
|
+
Measures latency (microseconds), throughput (GFLOPS), and speedup factor.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import sys
|
|
7
|
+
import io
|
|
8
|
+
import time
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
import numpy as np
|
|
11
|
+
|
|
12
|
+
# Ensure utf-8 stdout
|
|
13
|
+
if sys.platform == "win32":
|
|
14
|
+
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace")
|
|
15
|
+
|
|
16
|
+
# Add project root to sys.path
|
|
17
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
18
|
+
import nanogemm as ng
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def benchmark_size(dim: int, num_iters: int):
|
|
22
|
+
M = N = K = dim
|
|
23
|
+
A = np.random.randn(M, K).astype(np.float32)
|
|
24
|
+
B = np.random.randn(K, N).astype(np.float32)
|
|
25
|
+
C_out = np.empty((M, N), dtype=np.float32)
|
|
26
|
+
|
|
27
|
+
# Warmup
|
|
28
|
+
for _ in range(min(100, num_iters)):
|
|
29
|
+
_ = A @ B
|
|
30
|
+
_ = ng.matmul(A, B, out=C_out)
|
|
31
|
+
|
|
32
|
+
# Benchmark NumPy
|
|
33
|
+
t0 = time.perf_counter()
|
|
34
|
+
for _ in range(num_iters):
|
|
35
|
+
_ = A @ B
|
|
36
|
+
t_numpy = (time.perf_counter() - t0) / num_iters
|
|
37
|
+
|
|
38
|
+
# Benchmark NanoGEMM
|
|
39
|
+
t0 = time.perf_counter()
|
|
40
|
+
for _ in range(num_iters):
|
|
41
|
+
_ = ng.matmul(A, B, out=C_out)
|
|
42
|
+
t_nanogemm = (time.perf_counter() - t0) / num_iters
|
|
43
|
+
|
|
44
|
+
# Flops: 2 * M * N * K
|
|
45
|
+
flops = 2.0 * M * N * K
|
|
46
|
+
gflops_numpy = (flops / t_numpy) / 1e9
|
|
47
|
+
gflops_nanogemm = (flops / t_nanogemm) / 1e9
|
|
48
|
+
speedup = t_numpy / t_nanogemm
|
|
49
|
+
|
|
50
|
+
return {
|
|
51
|
+
"dim": f"{dim}x{dim}",
|
|
52
|
+
"numpy_us": t_numpy * 1e6,
|
|
53
|
+
"nanogemm_us": t_nanogemm * 1e6,
|
|
54
|
+
"numpy_gflops": gflops_numpy,
|
|
55
|
+
"nanogemm_gflops": gflops_nanogemm,
|
|
56
|
+
"speedup": speedup,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def main():
|
|
61
|
+
print("\n" + "=" * 82)
|
|
62
|
+
print(f" NanoGEMM Benchmark vs NumPy {np.__version__} on CPU")
|
|
63
|
+
print(f" Microkernel ISA: {ng.get_simd_isa()}")
|
|
64
|
+
print("=" * 82)
|
|
65
|
+
|
|
66
|
+
sizes = [
|
|
67
|
+
(16, 25000),
|
|
68
|
+
(32, 10000),
|
|
69
|
+
(64, 4000),
|
|
70
|
+
(128, 1000),
|
|
71
|
+
(256, 250),
|
|
72
|
+
(512, 50),
|
|
73
|
+
]
|
|
74
|
+
|
|
75
|
+
results = []
|
|
76
|
+
print(f"\n{'Matrix Size':<12} | {'NumPy (µs)':<12} | {'NanoGEMM (µs)':<14} | {'Speedup':<12} | {'NanoGEMM GFLOPS':<15}")
|
|
77
|
+
print("-" * 82)
|
|
78
|
+
|
|
79
|
+
for dim, iters in sizes:
|
|
80
|
+
res = benchmark_size(dim, iters)
|
|
81
|
+
results.append(res)
|
|
82
|
+
speedup_str = f"{res['speedup']:.2f}x"
|
|
83
|
+
if res['speedup'] >= 1.0:
|
|
84
|
+
speedup_str = f"{speedup_str} (FASTER)"
|
|
85
|
+
print(
|
|
86
|
+
f"{res['dim']:<12} | "
|
|
87
|
+
f"{res['numpy_us']:>10.2f} µs | "
|
|
88
|
+
f"{res['nanogemm_us']:>12.2f} µs | "
|
|
89
|
+
f"{speedup_str:<12} | "
|
|
90
|
+
f"{res['nanogemm_gflops']:>13.2f} GFLOPS"
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
print("-" * 82)
|
|
94
|
+
print(" Key Takeaway: NanoGEMM eliminates BLAS function overhead on small/medium tensors")
|
|
95
|
+
print("=" * 82 + "\n")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
if __name__ == "__main__":
|
|
99
|
+
main()
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""
|
|
2
|
+
NanoGEMM: Minimalist, bare-metal SIMD & Assembly Matrix Multiplication Engine.
|
|
3
|
+
Zero-overhead CPU microkernels for AI and scientific computing in Python.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import ctypes
|
|
9
|
+
import os
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
# Fast-Path: Native C-Extension Loader
|
|
18
|
+
# ---------------------------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
_HAS_C_EXT = False
|
|
21
|
+
try:
|
|
22
|
+
from nanogemm import _ext # type: ignore
|
|
23
|
+
_HAS_C_EXT = True
|
|
24
|
+
except ImportError:
|
|
25
|
+
try:
|
|
26
|
+
import _ext # type: ignore
|
|
27
|
+
_HAS_C_EXT = True
|
|
28
|
+
except ImportError:
|
|
29
|
+
_HAS_C_EXT = False
|
|
30
|
+
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
# Fallback: Dynamic Library (ctypes) Loader
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
def _find_library() -> Optional[str]:
|
|
36
|
+
pkg_dir = Path(__file__).resolve().parent
|
|
37
|
+
root_dir = pkg_dir.parent
|
|
38
|
+
|
|
39
|
+
candidates = []
|
|
40
|
+
if sys.platform.startswith("win"):
|
|
41
|
+
dll_name = "nanogemm.dll"
|
|
42
|
+
candidates = [
|
|
43
|
+
pkg_dir / dll_name,
|
|
44
|
+
root_dir / "build" / dll_name,
|
|
45
|
+
root_dir / dll_name,
|
|
46
|
+
]
|
|
47
|
+
elif sys.platform == "darwin":
|
|
48
|
+
dylib_name = "libnanogemm.dylib"
|
|
49
|
+
candidates = [
|
|
50
|
+
pkg_dir / dylib_name,
|
|
51
|
+
root_dir / "build" / dylib_name,
|
|
52
|
+
root_dir / dylib_name,
|
|
53
|
+
]
|
|
54
|
+
else:
|
|
55
|
+
so_name = "libnanogemm.so"
|
|
56
|
+
candidates = [
|
|
57
|
+
pkg_dir / so_name,
|
|
58
|
+
root_dir / "build" / so_name,
|
|
59
|
+
root_dir / so_name,
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
for p in candidates:
|
|
63
|
+
if p.is_file():
|
|
64
|
+
return str(p)
|
|
65
|
+
return None
|
|
66
|
+
|
|
67
|
+
_lib_path = _find_library()
|
|
68
|
+
_lib = ctypes.CDLL(_lib_path) if _lib_path else None
|
|
69
|
+
|
|
70
|
+
if _lib:
|
|
71
|
+
_c_float_p = ctypes.POINTER(ctypes.c_float)
|
|
72
|
+
_lib.nanogemm_sgemm.argtypes = [
|
|
73
|
+
ctypes.c_int, ctypes.c_int, ctypes.c_int,
|
|
74
|
+
ctypes.c_float,
|
|
75
|
+
ctypes.c_void_p, ctypes.c_int,
|
|
76
|
+
ctypes.c_void_p, ctypes.c_int,
|
|
77
|
+
ctypes.c_float,
|
|
78
|
+
ctypes.c_void_p, ctypes.c_int,
|
|
79
|
+
]
|
|
80
|
+
_lib.nanogemm_sgemm.restype = None
|
|
81
|
+
|
|
82
|
+
_lib.nanogemm_matmul.argtypes = [
|
|
83
|
+
ctypes.c_int, ctypes.c_int, ctypes.c_int,
|
|
84
|
+
ctypes.c_void_p, ctypes.c_void_p, ctypes.c_void_p,
|
|
85
|
+
]
|
|
86
|
+
_lib.nanogemm_matmul.restype = None
|
|
87
|
+
|
|
88
|
+
_lib.nanogemm_simd_isa.argtypes = []
|
|
89
|
+
_lib.nanogemm_simd_isa.restype = ctypes.c_char_p
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# ---------------------------------------------------------------------------
|
|
93
|
+
# Public Python API
|
|
94
|
+
# ---------------------------------------------------------------------------
|
|
95
|
+
|
|
96
|
+
def get_simd_isa() -> str:
|
|
97
|
+
"""Return the active SIMD instruction set in the compiled microkernel."""
|
|
98
|
+
if _HAS_C_EXT:
|
|
99
|
+
return _ext.simd_isa()
|
|
100
|
+
if _lib:
|
|
101
|
+
raw = _lib.nanogemm_simd_isa()
|
|
102
|
+
return raw.decode("utf-8") if raw else "Unknown"
|
|
103
|
+
return "Not compiled"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def matmul(
|
|
107
|
+
a: np.ndarray,
|
|
108
|
+
b: np.ndarray,
|
|
109
|
+
out: Optional[np.ndarray] = None,
|
|
110
|
+
) -> np.ndarray:
|
|
111
|
+
"""
|
|
112
|
+
Multiply two 2D matrices using hardware-accelerated SIMD GEMM.
|
|
113
|
+
|
|
114
|
+
Parameters
|
|
115
|
+
----------
|
|
116
|
+
a : np.ndarray
|
|
117
|
+
Matrix of shape (M, K). Must be 2-dimensional.
|
|
118
|
+
b : np.ndarray
|
|
119
|
+
Matrix of shape (K, N). Must be 2-dimensional.
|
|
120
|
+
out : np.ndarray, optional
|
|
121
|
+
Pre-allocated output buffer of shape (M, N) and dtype float32.
|
|
122
|
+
|
|
123
|
+
Returns
|
|
124
|
+
-------
|
|
125
|
+
np.ndarray
|
|
126
|
+
Result of matrix multiplication (M, N) with float32 dtype.
|
|
127
|
+
"""
|
|
128
|
+
if a.ndim != 2 or b.ndim != 2:
|
|
129
|
+
raise ValueError(
|
|
130
|
+
f"Expected 2D arrays, got a.ndim={a.ndim} and b.ndim={b.ndim}"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
M, K = a.shape
|
|
134
|
+
K_b, N = b.shape
|
|
135
|
+
|
|
136
|
+
if K != K_b:
|
|
137
|
+
raise ValueError(
|
|
138
|
+
f"Incompatible matrix dimensions: cannot multiply ({M}, {K}) by ({K_b}, {N})"
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
if not a.flags.c_contiguous or a.dtype != np.float32:
|
|
142
|
+
a = np.ascontiguousarray(a, dtype=np.float32)
|
|
143
|
+
if not b.flags.c_contiguous or b.dtype != np.float32:
|
|
144
|
+
b = np.ascontiguousarray(b, dtype=np.float32)
|
|
145
|
+
|
|
146
|
+
if out is None:
|
|
147
|
+
out = np.empty((M, N), dtype=np.float32)
|
|
148
|
+
else:
|
|
149
|
+
if out.shape != (M, N) or out.dtype != np.float32 or not out.flags.c_contiguous:
|
|
150
|
+
raise ValueError(
|
|
151
|
+
f"out must be contiguous float32 array of shape ({M}, {N})"
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
if _HAS_C_EXT:
|
|
155
|
+
_ext.matmul_fast(a, b, out)
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
if _lib:
|
|
159
|
+
_lib.nanogemm_matmul(M, N, K, a.ctypes.data, b.ctypes.data, out.ctypes.data)
|
|
160
|
+
return out
|
|
161
|
+
|
|
162
|
+
raise RuntimeError("NanoGEMM binary not available. Please compile nanogemm native binary.")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def sgemm(
|
|
166
|
+
a: np.ndarray,
|
|
167
|
+
b: np.ndarray,
|
|
168
|
+
alpha: float = 1.0,
|
|
169
|
+
beta: float = 0.0,
|
|
170
|
+
c: Optional[np.ndarray] = None,
|
|
171
|
+
) -> np.ndarray:
|
|
172
|
+
"""
|
|
173
|
+
Standard BLAS SGEMM: C = alpha * (A @ B) + beta * C.
|
|
174
|
+
"""
|
|
175
|
+
if a.ndim != 2 or b.ndim != 2:
|
|
176
|
+
raise ValueError(f"Expected 2D arrays, got {a.ndim} and {b.ndim}")
|
|
177
|
+
|
|
178
|
+
M, K = a.shape
|
|
179
|
+
K_b, N = b.shape
|
|
180
|
+
if K != K_b:
|
|
181
|
+
raise ValueError(f"Dimension mismatch: {a.shape} vs {b.shape}")
|
|
182
|
+
|
|
183
|
+
if not a.flags.c_contiguous or a.dtype != np.float32:
|
|
184
|
+
a = np.ascontiguousarray(a, dtype=np.float32)
|
|
185
|
+
if not b.flags.c_contiguous or b.dtype != np.float32:
|
|
186
|
+
b = np.ascontiguousarray(b, dtype=np.float32)
|
|
187
|
+
|
|
188
|
+
if c is None:
|
|
189
|
+
c = np.zeros((M, N), dtype=np.float32)
|
|
190
|
+
else:
|
|
191
|
+
if c.shape != (M, N) or c.dtype != np.float32 or not c.flags.c_contiguous:
|
|
192
|
+
raise ValueError(f"c must be contiguous float32 array of shape ({M}, {N})")
|
|
193
|
+
|
|
194
|
+
if _lib:
|
|
195
|
+
_lib.nanogemm_sgemm(M, N, K, float(alpha), a.ctypes.data, K, b.ctypes.data, N, float(beta), c.ctypes.data, N)
|
|
196
|
+
return c
|
|
197
|
+
|
|
198
|
+
# Fallback to matmul
|
|
199
|
+
res = matmul(a, b)
|
|
200
|
+
if beta == 0.0:
|
|
201
|
+
np.multiply(res, alpha, out=c)
|
|
202
|
+
else:
|
|
203
|
+
c[:] = alpha * res + beta * c
|
|
204
|
+
return c
|
|
Binary file
|