RamTorch 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ramtorch-0.2.2 → ramtorch-0.2.3}/PKG-INFO +12 -5
- {ramtorch-0.2.2 → ramtorch-0.2.3}/RamTorch.egg-info/PKG-INFO +12 -5
- ramtorch-0.2.3/RamTorch.egg-info/SOURCES.txt +29 -0
- ramtorch-0.2.3/RamTorch.egg-info/top_level.txt +6 -0
- ramtorch-0.2.3/env/bin/activate_this.py +36 -0
- ramtorch-0.2.3/kernels/int8_matmul.py +268 -0
- ramtorch-0.2.3/pyproject.toml +25 -0
- ramtorch-0.2.3/ramtorch/multi_gpu.py +767 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/stochastic_optimizers/adamw.py +1 -3
- ramtorch-0.2.3/setup.py +29 -0
- ramtorch-0.2.2/LICENSE +0 -201
- ramtorch-0.2.2/RamTorch.egg-info/SOURCES.txt +0 -15
- ramtorch-0.2.2/RamTorch.egg-info/top_level.txt +0 -1
- ramtorch-0.2.2/pyproject.toml +0 -19
- {ramtorch-0.2.2 → ramtorch-0.2.3}/README.md +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/RamTorch.egg-info/dependency_links.txt +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/__init__.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/helpers.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/modules/__init__.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/modules/linear.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/stochastic_optimizers/__init__.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/zero1.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/ramtorch/zero2.py +0 -0
- {ramtorch-0.2.2 → ramtorch-0.2.3}/setup.cfg +0 -0
|
@@ -1,16 +1,23 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: RamTorch
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: RAM is All You Need
|
|
5
|
+
Home-page: https://github.com/lodestone-rock/RamTorch
|
|
6
|
+
Author: Lodestone
|
|
5
7
|
Author-email: Lodestone <lodestone.rock@gmail.com>
|
|
6
|
-
License: Apache-2.0
|
|
7
8
|
Project-URL: Homepage, https://github.com/lodestone-rock/RamTorch
|
|
8
|
-
Classifier:
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Intended Audience :: Education
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
13
|
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
16
|
Requires-Python: >=3.8
|
|
11
17
|
Description-Content-Type: text/markdown
|
|
12
|
-
|
|
13
|
-
Dynamic:
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: home-page
|
|
20
|
+
Dynamic: requires-python
|
|
14
21
|
|
|
15
22
|
# RamTorch
|
|
16
23
|
|
|
@@ -1,16 +1,23 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: RamTorch
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: RAM is All You Need
|
|
5
|
+
Home-page: https://github.com/lodestone-rock/RamTorch
|
|
6
|
+
Author: Lodestone
|
|
5
7
|
Author-email: Lodestone <lodestone.rock@gmail.com>
|
|
6
|
-
License: Apache-2.0
|
|
7
8
|
Project-URL: Homepage, https://github.com/lodestone-rock/RamTorch
|
|
8
|
-
Classifier:
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Intended Audience :: Education
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
13
|
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
16
|
Requires-Python: >=3.8
|
|
11
17
|
Description-Content-Type: text/markdown
|
|
12
|
-
|
|
13
|
-
Dynamic:
|
|
18
|
+
Dynamic: author
|
|
19
|
+
Dynamic: home-page
|
|
20
|
+
Dynamic: requires-python
|
|
14
21
|
|
|
15
22
|
# RamTorch
|
|
16
23
|
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
setup.py
|
|
4
|
+
./env/bin/activate_this.py
|
|
5
|
+
./kernels/int8_matmul.py
|
|
6
|
+
./ramtorch/__init__.py
|
|
7
|
+
./ramtorch/helpers.py
|
|
8
|
+
./ramtorch/multi_gpu.py
|
|
9
|
+
./ramtorch/zero1.py
|
|
10
|
+
./ramtorch/zero2.py
|
|
11
|
+
./ramtorch/modules/__init__.py
|
|
12
|
+
./ramtorch/modules/linear.py
|
|
13
|
+
./ramtorch/stochastic_optimizers/__init__.py
|
|
14
|
+
./ramtorch/stochastic_optimizers/adamw.py
|
|
15
|
+
RamTorch.egg-info/PKG-INFO
|
|
16
|
+
RamTorch.egg-info/SOURCES.txt
|
|
17
|
+
RamTorch.egg-info/dependency_links.txt
|
|
18
|
+
RamTorch.egg-info/top_level.txt
|
|
19
|
+
env/bin/activate_this.py
|
|
20
|
+
kernels/int8_matmul.py
|
|
21
|
+
ramtorch/__init__.py
|
|
22
|
+
ramtorch/helpers.py
|
|
23
|
+
ramtorch/multi_gpu.py
|
|
24
|
+
ramtorch/zero1.py
|
|
25
|
+
ramtorch/zero2.py
|
|
26
|
+
ramtorch/modules/__init__.py
|
|
27
|
+
ramtorch/modules/linear.py
|
|
28
|
+
ramtorch/stochastic_optimizers/__init__.py
|
|
29
|
+
ramtorch/stochastic_optimizers/adamw.py
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Activate virtualenv for current interpreter:
|
|
3
|
+
|
|
4
|
+
Use exec(open(this_file).read(), {'__file__': this_file}).
|
|
5
|
+
|
|
6
|
+
This can be used when you must use an existing Python interpreter, not the virtualenv bin/python.
|
|
7
|
+
""" # noqa: D415
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import site
|
|
12
|
+
import sys
|
|
13
|
+
|
|
14
|
+
try:
|
|
15
|
+
abs_file = os.path.abspath(__file__)
|
|
16
|
+
except NameError as exc:
|
|
17
|
+
msg = "You must use exec(open(this_file).read(), {'__file__': this_file})"
|
|
18
|
+
raise AssertionError(msg) from exc
|
|
19
|
+
|
|
20
|
+
bin_dir = os.path.dirname(abs_file)
|
|
21
|
+
base = bin_dir[: -len('bin') - 1] # strip away the bin part from the __file__, plus the path separator
|
|
22
|
+
|
|
23
|
+
# prepend bin to PATH (this file is inside the bin directory)
|
|
24
|
+
os.environ["PATH"] = os.pathsep.join([bin_dir, *os.environ.get("PATH", "").split(os.pathsep)])
|
|
25
|
+
os.environ["VIRTUAL_ENV"] = base # virtual env is right above bin directory
|
|
26
|
+
os.environ["VIRTUAL_ENV_PROMPT"] = '' or os.path.basename(base)
|
|
27
|
+
|
|
28
|
+
# add the virtual environments libraries to the host python import mechanism
|
|
29
|
+
prev_length = len(sys.path)
|
|
30
|
+
for lib in '../lib/python3.12/site-packages'.split(os.pathsep):
|
|
31
|
+
path = os.path.realpath(os.path.join(bin_dir, lib))
|
|
32
|
+
site.addsitedir(path.decode("utf-8") if '' else path)
|
|
33
|
+
sys.path[:] = sys.path[prev_length:] + sys.path[0:prev_length]
|
|
34
|
+
|
|
35
|
+
sys.real_prefix = sys.prefix
|
|
36
|
+
sys.prefix = base
|
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
import torch
|
|
2
|
+
import triton
|
|
3
|
+
import triton.language as tl
|
|
4
|
+
from triton import Config
|
|
5
|
+
from typing import Tuple
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
"""
|
|
9
|
+
simplified explanation of the scaled int8 matmul algorithm
|
|
10
|
+
adopted from deepseek scaled FP8 matmul and jetfire paper
|
|
11
|
+
https://arxiv.org/abs/2403.12422
|
|
12
|
+
https://github.com/deepseek-ai/DeepSeek-V3/blob/main/inference/kernel.py
|
|
13
|
+
|
|
14
|
+
N dimension →
|
|
15
|
+
INT8 weights scaler per block
|
|
16
|
+
┌-----┬-----┬─────┬─────┐ ┌-----┬-----┬─────┬─────┐
|
|
17
|
+
: b00 : b01 : b02 | b03 | : : : | |
|
|
18
|
+
├-----┼-----┼─────┼─────┤ :b_s00:b_s10:b_s20|b_s30|
|
|
19
|
+
K : b10 : b11 : b12 | b13 | : : : | |
|
|
20
|
+
dim ├-----┼-----┼─────┼─────┤ ├-----┼-----┼─────┼─────┤
|
|
21
|
+
↓ | b20 | b21 | b22 | b23 | | | | | |
|
|
22
|
+
├─────┼─────┼─────┼─────┤ |b_s01|b_s11|b_s21|b_s31|
|
|
23
|
+
| b30 | b31 | b32 | b33 | | | | | |
|
|
24
|
+
└─────┴─────┴─────┴─────┘ └─────┴─────┴─────┴─────┘
|
|
25
|
+
┌-----┬-----┐
|
|
26
|
+
: b00 : b01 :
|
|
27
|
+
├─── blk ───┤ ├-----┼-----┤
|
|
28
|
+
: b10 : b11 :
|
|
29
|
+
K dimension → └-----┴-----┘
|
|
30
|
+
INT8 activations
|
|
31
|
+
┌-----┬-----┬─────┬─────┐ ┌-----┬-----┐ ┌-----┬-----┐ ┌-----------┐ ┌-----┬-----┐ ┌-----┬-----┐
|
|
32
|
+
: a00 : a01 : a02 | a03 | : a00 : a01 : : @ : @ : : a_s00 : : : : :acc00:acc01:
|
|
33
|
+
├-----┼-----┼─────┼─────┤ ├-----┼-----┤ ├-----┼-----┤ * ├-----------┤ * :b_s00:b_s10: = ├-----┼-----┤
|
|
34
|
+
M : a10 : a11 : a12 | a13 | : a10 : a11 : : @ : @ : : a_s10 : : : : :acc10:acc11:
|
|
35
|
+
dim ├-----┼-----┼─────┼─────┤ └-----┴-----┘ └-----┴-----┘ └-----------┘ └-----┴-----┘ └-----┴-----┘
|
|
36
|
+
↓ | a20 | a21 | a22 | a23 | INT8 matmul acc in INT32 rescale the FP32 intermediate accumulate
|
|
37
|
+
├─────┼─────┼─────┼─────┤ then cast to FP32 "rank 1" hadamard scaler intermediate
|
|
38
|
+
| a30 | a31 | a32 | a33 |
|
|
39
|
+
└─────┴─────┴─────┴─────┘
|
|
40
|
+
scaler per block
|
|
41
|
+
┌-----------┬───────────┐
|
|
42
|
+
: a_s00 : a_s01 |
|
|
43
|
+
├-----------┼───────────┤
|
|
44
|
+
: a_s10 : a_s11 |
|
|
45
|
+
├-----------┼───────────┤
|
|
46
|
+
| a_s20 | a_s21 |
|
|
47
|
+
├───────────┼───────────┤
|
|
48
|
+
| a_s30 | a_s31 |
|
|
49
|
+
└───────────┴───────────┘
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@triton.jit
|
|
54
|
+
def act_quant_kernel(x_ptr, y_ptr, s_ptr, BLOCK_SIZE: tl.constexpr):
|
|
55
|
+
"""
|
|
56
|
+
Quantizes the input tensor `x_ptr` and stores the result in `y_ptr` and the scaling factor in `s_ptr`.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
x_ptr (triton.Pointer): Pointer to the input tensor.
|
|
60
|
+
y_ptr (triton.Pointer): Pointer to the output tensor where quantized values will be stored.
|
|
61
|
+
s_ptr (triton.Pointer): Pointer to the output tensor where scaling factors will be stored.
|
|
62
|
+
BLOCK_SIZE (tl.constexpr): The size of the block to be processed by each program instance.
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
None
|
|
66
|
+
"""
|
|
67
|
+
pid = tl.program_id(axis=0)
|
|
68
|
+
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
|
|
69
|
+
x = tl.load(x_ptr + offs).to(tl.float32)
|
|
70
|
+
amax = tl.max(tl.abs(x)) # reduction
|
|
71
|
+
# amax = tl.maximum(amax, 1e-4) # clamp to 1e-4
|
|
72
|
+
s = amax / 127.0
|
|
73
|
+
y = x / s
|
|
74
|
+
y = y.to(y_ptr.dtype.element_ty)
|
|
75
|
+
tl.store(y_ptr + offs, y)
|
|
76
|
+
tl.store(s_ptr + pid, s)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def act_quant(
|
|
80
|
+
x: torch.Tensor, block_size: int = 128
|
|
81
|
+
) -> Tuple[torch.Tensor, torch.Tensor]:
|
|
82
|
+
"""
|
|
83
|
+
Quantizes the input tensor `x` using block-wise quantization.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
x (torch.Tensor): The input tensor to be quantized. Must be contiguous and its last dimension size must be divisible by `block_size`.
|
|
87
|
+
block_size (int, optional): The size of the blocks to be used for quantization. Default is 128.
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
Tuple[torch.Tensor, torch.Tensor]: A tuple containing:
|
|
91
|
+
- The quantized tensor with dtype `torch.int8`.
|
|
92
|
+
- A tensor of scaling factors with dtype `torch.float32`.
|
|
93
|
+
"""
|
|
94
|
+
assert x.is_contiguous(), "Input tensor must be contiguous"
|
|
95
|
+
assert (
|
|
96
|
+
x.size(-1) % block_size == 0
|
|
97
|
+
), f"Last dimension size must be divisible by block_size (block_size={block_size})"
|
|
98
|
+
y = torch.empty_like(x, dtype=torch.int8)
|
|
99
|
+
s = x.new_empty(*x.size()[:-1], x.size(-1) // block_size, dtype=torch.float32)
|
|
100
|
+
grid = lambda meta: (triton.cdiv(x.numel(), meta["BLOCK_SIZE"]),)
|
|
101
|
+
act_quant_kernel[grid](x, y, s, BLOCK_SIZE=block_size)
|
|
102
|
+
return y, s
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@triton.jit
|
|
106
|
+
def weight_dequant_kernel(x_ptr, s_ptr, y_ptr, M, N, BLOCK_SIZE: tl.constexpr):
|
|
107
|
+
"""
|
|
108
|
+
Dequantizes weights using the provided scaling factors and stores the result.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
x_ptr (tl.pointer): Pointer to the quantized weights.
|
|
112
|
+
s_ptr (tl.pointer): Pointer to the scaling factors.
|
|
113
|
+
y_ptr (tl.pointer): Pointer to the output buffer for dequantized weights.
|
|
114
|
+
M (int): Number of rows in the weight matrix.
|
|
115
|
+
N (int): Number of columns in the weight matrix.
|
|
116
|
+
BLOCK_SIZE (tl.constexpr): Size of the block for tiling.
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
None
|
|
120
|
+
"""
|
|
121
|
+
pid_m = tl.program_id(axis=0)
|
|
122
|
+
pid_n = tl.program_id(axis=1)
|
|
123
|
+
n = tl.cdiv(N, BLOCK_SIZE)
|
|
124
|
+
offs_m = pid_m * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
|
|
125
|
+
offs_n = pid_n * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
|
|
126
|
+
offs = offs_m[:, None] * N + offs_n[None, :]
|
|
127
|
+
mask = (offs_m[:, None] < M) & (offs_n[None, :] < N)
|
|
128
|
+
x = tl.load(x_ptr + offs, mask=mask).to(tl.float32)
|
|
129
|
+
s = tl.load(s_ptr + pid_m * n + pid_n)
|
|
130
|
+
y = x * s
|
|
131
|
+
tl.store(y_ptr + offs, y, mask=mask)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def weight_dequant(
|
|
135
|
+
x: torch.Tensor, s: torch.Tensor, block_size: int = 128
|
|
136
|
+
) -> torch.Tensor:
|
|
137
|
+
"""
|
|
138
|
+
Dequantizes the given weight tensor using the provided scale tensor.
|
|
139
|
+
|
|
140
|
+
Args:
|
|
141
|
+
x (torch.Tensor): The quantized weight tensor of shape (M, N).
|
|
142
|
+
s (torch.Tensor): The scale tensor of shape (M//block_size, N//block_size).
|
|
143
|
+
block_size (int, optional): The block size to use for dequantization. Defaults to 128.
|
|
144
|
+
|
|
145
|
+
Returns:
|
|
146
|
+
torch.Tensor: The dequantized weight tensor of the same shape as `x`.
|
|
147
|
+
|
|
148
|
+
Raises:
|
|
149
|
+
AssertionError: If `x` or `s` are not contiguous or if their dimensions are not 2.
|
|
150
|
+
"""
|
|
151
|
+
assert x.is_contiguous() and s.is_contiguous(), "Input tensors must be contiguous"
|
|
152
|
+
assert x.dim() == 2 and s.dim() == 2, "Input tensors must have 2 dimensions"
|
|
153
|
+
M, N = x.size()
|
|
154
|
+
y = torch.empty_like(x, dtype=torch.get_default_dtype())
|
|
155
|
+
grid = lambda meta: (
|
|
156
|
+
triton.cdiv(M, meta["BLOCK_SIZE"]),
|
|
157
|
+
triton.cdiv(N, meta["BLOCK_SIZE"]),
|
|
158
|
+
)
|
|
159
|
+
weight_dequant_kernel[grid](x, s, y, M, N, BLOCK_SIZE=block_size)
|
|
160
|
+
return y
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# matmul intermediate block size is hardcoded to 128
|
|
164
|
+
int8_gemm_configs = [
|
|
165
|
+
Config(
|
|
166
|
+
{"BLOCK_SIZE_M": block_m, "BLOCK_SIZE_N": block_n, "BLOCK_SIZE_K": 128},
|
|
167
|
+
num_stages=num_stages,
|
|
168
|
+
num_warps=8,
|
|
169
|
+
)
|
|
170
|
+
for block_m in [16, 32, 64]
|
|
171
|
+
for block_n in [32, 64, 128]
|
|
172
|
+
for num_stages in [3, 4, 5, 6]
|
|
173
|
+
]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
@triton.autotune(configs=int8_gemm_configs, key=["N", "K"])
|
|
177
|
+
@triton.jit
|
|
178
|
+
def int8_gemm_kernel(
|
|
179
|
+
a_ptr,
|
|
180
|
+
b_ptr,
|
|
181
|
+
c_ptr,
|
|
182
|
+
a_s_ptr,
|
|
183
|
+
b_s_ptr,
|
|
184
|
+
M,
|
|
185
|
+
N: tl.constexpr,
|
|
186
|
+
K: tl.constexpr,
|
|
187
|
+
BLOCK_SIZE_M: tl.constexpr,
|
|
188
|
+
BLOCK_SIZE_N: tl.constexpr,
|
|
189
|
+
BLOCK_SIZE_K: tl.constexpr,
|
|
190
|
+
):
|
|
191
|
+
"""
|
|
192
|
+
Performs a matrix multiplication operation on INT8 matrices with scaling factors.
|
|
193
|
+
|
|
194
|
+
Args:
|
|
195
|
+
a_ptr (tl.tensor): Pointer to the first input matrix A.
|
|
196
|
+
b_ptr (tl.tensor): Pointer to the second input matrix B.
|
|
197
|
+
c_ptr (tl.tensor): Pointer to the output matrix C.
|
|
198
|
+
a_s_ptr (tl.tensor): Pointer to the scaling factors for matrix A.
|
|
199
|
+
b_s_ptr (tl.tensor): Pointer to the scaling factors for matrix B.
|
|
200
|
+
M (int): Number of rows in matrix A and C.
|
|
201
|
+
N (tl.constexpr): Number of columns in matrix B and C.
|
|
202
|
+
K (tl.constexpr): Number of columns in matrix A and rows in matrix B.
|
|
203
|
+
BLOCK_SIZE_M (tl.constexpr): Block size for the M dimension.
|
|
204
|
+
BLOCK_SIZE_N (tl.constexpr): Block size for the N dimension.
|
|
205
|
+
BLOCK_SIZE_K (tl.constexpr): Block size for the K dimension.
|
|
206
|
+
|
|
207
|
+
Returns:
|
|
208
|
+
None
|
|
209
|
+
"""
|
|
210
|
+
pid_m = tl.program_id(axis=0)
|
|
211
|
+
pid_n = tl.program_id(axis=1)
|
|
212
|
+
k = tl.cdiv(K, BLOCK_SIZE_K)
|
|
213
|
+
offs_m = (pid_m * BLOCK_SIZE_M + tl.arange(0, BLOCK_SIZE_M)) % M
|
|
214
|
+
offs_n = (pid_n * BLOCK_SIZE_N + tl.arange(0, BLOCK_SIZE_N)) % N
|
|
215
|
+
offs_k = tl.arange(0, BLOCK_SIZE_K)
|
|
216
|
+
a_ptrs = a_ptr + offs_m[:, None] * K + offs_k[None, :]
|
|
217
|
+
b_ptrs = b_ptr + offs_n[None, :] * K + offs_k[:, None]
|
|
218
|
+
a_s_ptrs = a_s_ptr + offs_m * k
|
|
219
|
+
b_s_ptrs = b_s_ptr + offs_n * k
|
|
220
|
+
|
|
221
|
+
accumulator = tl.zeros((BLOCK_SIZE_M, BLOCK_SIZE_N), dtype=tl.float32)
|
|
222
|
+
for i in range(k):
|
|
223
|
+
a = tl.load(a_ptrs, mask=offs_k[None, :] < K - i * BLOCK_SIZE_K, other=0.0)
|
|
224
|
+
b = tl.load(b_ptrs, mask=offs_k[:, None] < K - i * BLOCK_SIZE_K, other=0.0)
|
|
225
|
+
a_s = tl.load(a_s_ptrs)
|
|
226
|
+
b_s = tl.load(b_s_ptrs)
|
|
227
|
+
# Cast to float32 before multiplying with scaling factors.
|
|
228
|
+
dot_prod = tl.dot(a, b)
|
|
229
|
+
accumulator += dot_prod.to(tl.float32) * a_s[:, None] * b_s[None, :]
|
|
230
|
+
a_ptrs += BLOCK_SIZE_K
|
|
231
|
+
b_ptrs += BLOCK_SIZE_K
|
|
232
|
+
a_s_ptrs += 1
|
|
233
|
+
b_s_ptrs += 1
|
|
234
|
+
c = accumulator.to(c_ptr.dtype.element_ty)
|
|
235
|
+
offs_m = pid_m * BLOCK_SIZE_M + tl.arange(0, BLOCK_SIZE_M)
|
|
236
|
+
offs_n = pid_n * BLOCK_SIZE_N + tl.arange(0, BLOCK_SIZE_N)
|
|
237
|
+
c_ptrs = c_ptr + offs_m[:, None] * N + offs_n[None, :]
|
|
238
|
+
mask = (offs_m[:, None] < M) & (offs_n[None, :] < N)
|
|
239
|
+
tl.store(c_ptrs, c, mask=mask)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def int8_gemm(a: torch.Tensor, a_s: torch.Tensor, b: torch.Tensor, b_s: torch.Tensor):
|
|
243
|
+
"""
|
|
244
|
+
Perform a matrix multiplication using INT8 precision.
|
|
245
|
+
|
|
246
|
+
Args:
|
|
247
|
+
a (torch.Tensor): The first input matrix, must be contiguous.
|
|
248
|
+
a_s (torch.Tensor): The scaling factor for the first input matrix, must be contiguous.
|
|
249
|
+
b (torch.Tensor): The second input matrix, must be contiguous.
|
|
250
|
+
b_s (torch.Tensor): The scaling factor for the second input matrix, must be contiguous.
|
|
251
|
+
|
|
252
|
+
Returns:
|
|
253
|
+
torch.Tensor: The result of the matrix multiplication.
|
|
254
|
+
"""
|
|
255
|
+
assert a.is_contiguous() and b.is_contiguous(), "Input tensors must be contiguous"
|
|
256
|
+
assert (
|
|
257
|
+
a_s.is_contiguous() and b_s.is_contiguous()
|
|
258
|
+
), "Scaling factor tensors must be contiguous"
|
|
259
|
+
K = a.size(-1)
|
|
260
|
+
M = a.numel() // K
|
|
261
|
+
N = b.size(0)
|
|
262
|
+
c = a.new_empty(*a.size()[:-1], N, dtype=torch.get_default_dtype())
|
|
263
|
+
grid = lambda META: (
|
|
264
|
+
triton.cdiv(M, META["BLOCK_SIZE_M"]),
|
|
265
|
+
triton.cdiv(N, META["BLOCK_SIZE_N"]),
|
|
266
|
+
)
|
|
267
|
+
int8_gemm_kernel[grid](a, b, c, a_s, b_s, M, N, K)
|
|
268
|
+
return c
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "RamTorch"
|
|
3
|
+
version = "0.2.3"
|
|
4
|
+
authors = [{name = "Lodestone", email = "lodestone.rock@gmail.com"}]
|
|
5
|
+
description = "RAM is All You Need"
|
|
6
|
+
readme = "README.md"
|
|
7
|
+
requires-python = ">=3.8"
|
|
8
|
+
classifiers = [
|
|
9
|
+
"Development Status :: 4 - Beta",
|
|
10
|
+
"Intended Audience :: Developers",
|
|
11
|
+
"Intended Audience :: Education",
|
|
12
|
+
"Intended Audience :: Science/Research",
|
|
13
|
+
"Operating System :: OS Independent",
|
|
14
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.urls]
|
|
19
|
+
Homepage = "https://github.com/lodestone-rock/RamTorch"
|
|
20
|
+
|
|
21
|
+
[tool.setuptools.packages.find]
|
|
22
|
+
exclude = ["examples*"]
|
|
23
|
+
|
|
24
|
+
[tool.setuptools]
|
|
25
|
+
license-files = []
|