piper-kernels 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ """Reusable PyTorch inference operators and optimized kernels."""
@@ -0,0 +1 @@
1
+ """Optimized attention operators."""
@@ -0,0 +1 @@
1
+ """Optional optimized backends for attention operators."""
@@ -0,0 +1,5 @@
1
+ """Rotated quantized weights with transparent PyTorch linear dispatch."""
2
+
3
+ from ._int8.tensor import ConvRotInt8Tensor
4
+
5
+ __all__ = ["ConvRotInt8Tensor"]
@@ -0,0 +1 @@
1
+ """Internal INT8 ConvRot implementation."""
@@ -0,0 +1 @@
1
+ """Optional optimized backends for INT8 ConvRot operators."""
@@ -0,0 +1,357 @@
1
+ """Triton backend for rotated INT8 W8A8 linear layers."""
2
+
3
+ # Triton's JIT launcher accepts compile-time options not represented in its
4
+ # Python call signature.
5
+ # pyright: reportCallIssue=false
6
+
7
+ import torch
8
+ import triton
9
+ import triton.language as tl
10
+ from triton.language.extra import libdevice
11
+
12
+
13
+ @triton.jit
14
+ def _hadamard_stage(values, offsets, stride: tl.constexpr):
15
+ """Apply one H4 Kronecker factor to a flattened regular Hadamard row."""
16
+ digit = (offsets // stride) % 4
17
+ base = offsets - digit * stride
18
+ a = tl.gather(values, base, 0)
19
+ b = tl.gather(values, base + stride, 0)
20
+ c = tl.gather(values, base + 2 * stride, 0)
21
+ d = tl.gather(values, base + 3 * stride, 0)
22
+ return tl.where(
23
+ digit == 0,
24
+ a + b + c - d,
25
+ tl.where(
26
+ digit == 1,
27
+ a + b - c + d,
28
+ tl.where(digit == 2, a - b + c + d, -a + b + c + d),
29
+ ),
30
+ )
31
+
32
+
33
+ @triton.jit
34
+ def _normalize_for_int8(values, scale, logical_dtype_code: tl.constexpr):
35
+ """Normalize values without dividing by an underflowed logical scale."""
36
+ if logical_dtype_code == 1:
37
+ logical_scale = scale.to(tl.float16)
38
+ safe_logical_scale = tl.where(logical_scale == 0, 1.0, logical_scale).to(tl.float16)
39
+ scaled = (values / safe_logical_scale).to(tl.float16)
40
+ return tl.where(
41
+ logical_scale == 0,
42
+ values.to(tl.float32) / scale,
43
+ scaled.to(tl.float32),
44
+ )
45
+ elif logical_dtype_code == 2:
46
+ return (values / scale.to(tl.bfloat16)).to(tl.bfloat16)
47
+ else:
48
+ return values / scale
49
+
50
+
51
+ @triton.jit
52
+ def _rotate_groups_kernel(
53
+ x_ptr,
54
+ out_ptr,
55
+ row_width,
56
+ groups_per_row,
57
+ group_size: tl.constexpr,
58
+ inverse_sqrt_group: tl.constexpr,
59
+ ):
60
+ group_id = tl.program_id(0)
61
+ row = group_id // groups_per_row
62
+ group = group_id % groups_per_row
63
+ offsets = tl.arange(0, group_size)
64
+ pointers = x_ptr + row * row_width + group * group_size + offsets
65
+ values = tl.load(pointers).to(tl.float32)
66
+
67
+ values = _hadamard_stage(values, offsets, 1)
68
+ values = _hadamard_stage(values, offsets, 4)
69
+ if group_size >= 64:
70
+ values = _hadamard_stage(values, offsets, 16)
71
+ if group_size >= 256:
72
+ values = _hadamard_stage(values, offsets, 64)
73
+
74
+ tl.store(out_ptr + row * row_width + group * group_size + offsets, values * inverse_sqrt_group)
75
+
76
+
77
+ @triton.jit
78
+ def _quantize_rows_kernel(
79
+ x_ptr,
80
+ q_ptr,
81
+ scale_ptr,
82
+ row_width,
83
+ block_size: tl.constexpr,
84
+ input_dtype_code: tl.constexpr,
85
+ ):
86
+ row = tl.program_id(0)
87
+ offsets = tl.arange(0, block_size)
88
+ mask = offsets < row_width
89
+ values = tl.load(x_ptr + row * row_width + offsets, mask=mask, other=0.0)
90
+ scale = tl.maximum(tl.max(tl.abs(values), axis=0) / 127.0, 1e-30)
91
+ scaled = _normalize_for_int8(values, scale, input_dtype_code)
92
+ quantized = tl.clamp(libdevice.rint(scaled.to(tl.float32)), -128.0, 127.0).to(tl.int8)
93
+ tl.store(q_ptr + row * row_width + offsets, quantized, mask=mask)
94
+ tl.store(scale_ptr + row, scale)
95
+
96
+
97
+ @triton.jit
98
+ def _requantize_addmm_rows_kernel(
99
+ q_ptr,
100
+ scale_ptr,
101
+ update_ptr,
102
+ row_width,
103
+ stride_q_row,
104
+ stride_q_col,
105
+ stride_scale_row,
106
+ stride_update_row,
107
+ stride_update_col,
108
+ beta,
109
+ alpha,
110
+ block_size: tl.constexpr,
111
+ logical_dtype_code: tl.constexpr,
112
+ has_base: tl.constexpr,
113
+ has_update: tl.constexpr,
114
+ ):
115
+ row = tl.program_id(0)
116
+ offsets = tl.arange(0, block_size)
117
+ mask = offsets < row_width
118
+ if has_base:
119
+ quantized = tl.load(
120
+ q_ptr + row * stride_q_row + offsets * stride_q_col,
121
+ mask=mask,
122
+ other=0,
123
+ )
124
+ old_scale = tl.load(scale_ptr + row * stride_scale_row)
125
+ values = beta * quantized.to(tl.float32) * old_scale
126
+ else:
127
+ values = tl.zeros((block_size,), dtype=tl.float32)
128
+ if has_update:
129
+ update = tl.load(
130
+ update_ptr + row * stride_update_row + offsets * stride_update_col,
131
+ mask=mask,
132
+ other=0.0,
133
+ )
134
+ values += alpha * update.to(tl.float32)
135
+ if logical_dtype_code == 1:
136
+ values = values.to(tl.float16)
137
+ elif logical_dtype_code == 2:
138
+ values = values.to(tl.bfloat16)
139
+ scale = tl.maximum(tl.max(tl.abs(values).to(tl.float32), axis=0) / 127.0, 1e-30)
140
+ scaled = _normalize_for_int8(values, scale, logical_dtype_code)
141
+ quantized = tl.clamp(libdevice.rint(scaled.to(tl.float32)), -128.0, 127.0).to(tl.int8)
142
+ tl.store(
143
+ q_ptr + row * stride_q_row + offsets * stride_q_col,
144
+ quantized,
145
+ mask=mask,
146
+ )
147
+ tl.store(scale_ptr + row * stride_scale_row, scale)
148
+
149
+
150
+ @triton.jit
151
+ def _int8_matmul_kernel(
152
+ activation_ptr,
153
+ weight_ptr,
154
+ output_ptr,
155
+ activation_scale_ptr,
156
+ weight_scale_ptr,
157
+ bias_ptr,
158
+ m,
159
+ n,
160
+ k,
161
+ stride_am,
162
+ stride_ak,
163
+ stride_wn,
164
+ stride_wk,
165
+ stride_om,
166
+ stride_on,
167
+ block_m: tl.constexpr,
168
+ block_n: tl.constexpr,
169
+ block_k: tl.constexpr,
170
+ has_bias: tl.constexpr,
171
+ ):
172
+ pid_m = tl.program_id(0)
173
+ pid_n = tl.program_id(1)
174
+ offsets_m = pid_m * block_m + tl.arange(0, block_m)
175
+ offsets_n = pid_n * block_n + tl.arange(0, block_n)
176
+ offsets_k = tl.arange(0, block_k)
177
+
178
+ activation_pointers = (
179
+ activation_ptr + offsets_m[:, None] * stride_am + offsets_k[None, :] * stride_ak
180
+ )
181
+ weight_pointers = weight_ptr + offsets_n[None, :] * stride_wn + offsets_k[:, None] * stride_wk
182
+ accumulator = tl.zeros((block_m, block_n), dtype=tl.int32)
183
+
184
+ for k_offset in range(tl.cdiv(k, block_k)):
185
+ activation = tl.load(
186
+ activation_pointers,
187
+ mask=(offsets_m[:, None] < m) & (offsets_k[None, :] < k - k_offset * block_k),
188
+ other=0,
189
+ )
190
+ weight = tl.load(
191
+ weight_pointers,
192
+ mask=(offsets_n[None, :] < n) & (offsets_k[:, None] < k - k_offset * block_k),
193
+ other=0,
194
+ )
195
+ accumulator += tl.dot(activation, weight)
196
+ activation_pointers += block_k * stride_ak
197
+ weight_pointers += block_k * stride_wk
198
+
199
+ activation_scale = tl.load(activation_scale_ptr + offsets_m, mask=offsets_m < m, other=0.0)
200
+ weight_scale = tl.load(weight_scale_ptr + offsets_n, mask=offsets_n < n, other=0.0)
201
+ result = accumulator.to(tl.float32) * activation_scale[:, None] * weight_scale[None, :]
202
+ if has_bias:
203
+ bias = tl.load(bias_ptr + offsets_n, mask=offsets_n < n, other=0.0)
204
+ result += bias[None, :]
205
+
206
+ output_pointers = output_ptr + offsets_m[:, None] * stride_om + offsets_n[None, :] * stride_on
207
+ tl.store(
208
+ output_pointers,
209
+ result,
210
+ mask=(offsets_m[:, None] < m) & (offsets_n[None, :] < n),
211
+ )
212
+
213
+
214
+ @torch.library.custom_op("piper_kernels::convrot_int8_linear", mutates_args=())
215
+ def triton_convrot_int8_linear(
216
+ activation: torch.Tensor,
217
+ weight: torch.Tensor,
218
+ weight_scale: torch.Tensor,
219
+ bias: torch.Tensor | None,
220
+ group_size: int,
221
+ ) -> torch.Tensor:
222
+ """Run ConvRot activation rotation, dynamic quantization, and INT8 GEMM."""
223
+ original_shape = activation.shape
224
+ activation_2d = activation.reshape(-1, original_shape[-1]).contiguous()
225
+ m, k = activation_2d.shape
226
+ n = weight.shape[0]
227
+ if m == 0 or n == 0:
228
+ return activation.new_empty((*original_shape[:-1], n))
229
+
230
+ groups_per_row = k // group_size
231
+ rotated = torch.empty_like(activation_2d)
232
+ _rotate_groups_kernel[(m * groups_per_row,)](
233
+ activation_2d,
234
+ rotated,
235
+ k,
236
+ groups_per_row,
237
+ group_size=group_size,
238
+ inverse_sqrt_group=group_size**-0.5,
239
+ num_warps=4,
240
+ )
241
+
242
+ quantized = torch.empty_like(rotated, dtype=torch.int8)
243
+ activation_scale = torch.empty(m, device=activation.device, dtype=torch.float32)
244
+ quant_block = max(128, triton.next_power_of_2(k))
245
+ if activation.dtype is torch.float16:
246
+ input_dtype_code = 1
247
+ elif activation.dtype is torch.bfloat16:
248
+ input_dtype_code = 2
249
+ else:
250
+ input_dtype_code = 0
251
+ _quantize_rows_kernel[(m,)](
252
+ rotated,
253
+ quantized,
254
+ activation_scale,
255
+ k,
256
+ block_size=quant_block,
257
+ input_dtype_code=input_dtype_code,
258
+ num_warps=8,
259
+ )
260
+
261
+ output = torch.empty((m, n), device=activation.device, dtype=activation.dtype)
262
+ block_m = 32 if m < 64 else 64
263
+ block_n = 64 if n < 128 else 128
264
+ block_k = 32
265
+ grid = (triton.cdiv(m, block_m), triton.cdiv(n, block_n))
266
+ bias_pointer = bias if bias is not None else activation
267
+ _int8_matmul_kernel[grid](
268
+ quantized,
269
+ weight,
270
+ output,
271
+ activation_scale,
272
+ weight_scale,
273
+ bias_pointer,
274
+ m,
275
+ n,
276
+ k,
277
+ quantized.stride(0),
278
+ quantized.stride(1),
279
+ weight.stride(0),
280
+ weight.stride(1),
281
+ output.stride(0),
282
+ output.stride(1),
283
+ block_m=block_m,
284
+ block_n=block_n,
285
+ block_k=block_k,
286
+ has_bias=bias is not None,
287
+ num_stages=3,
288
+ num_warps=4,
289
+ )
290
+ return output.reshape(*original_shape[:-1], n)
291
+
292
+
293
+ def triton_convrot_int8_addmm_(
294
+ qdata: torch.Tensor,
295
+ scale: torch.Tensor,
296
+ mat1: torch.Tensor,
297
+ mat2: torch.Tensor,
298
+ group_size: int,
299
+ beta: float,
300
+ alpha: float,
301
+ ) -> None:
302
+ """Apply an addmm update in the rotated basis and requantize the weight in place."""
303
+ out_features, in_features = qdata.shape
304
+ has_update = alpha != 0 and mat1.shape[1] != 0
305
+ if has_update:
306
+ mat2_contiguous = mat2.contiguous()
307
+ rotated_mat2 = torch.empty_like(mat2_contiguous)
308
+ groups_per_row = in_features // group_size
309
+ _rotate_groups_kernel[(mat2.shape[0] * groups_per_row,)](
310
+ mat2_contiguous,
311
+ rotated_mat2,
312
+ in_features,
313
+ groups_per_row,
314
+ group_size=group_size,
315
+ inverse_sqrt_group=group_size**-0.5,
316
+ num_warps=4,
317
+ )
318
+ update = torch.mm(mat1, rotated_mat2)
319
+ else:
320
+ update = qdata
321
+
322
+ if mat1.dtype is torch.float16:
323
+ logical_dtype_code = 1
324
+ elif mat1.dtype is torch.bfloat16:
325
+ logical_dtype_code = 2
326
+ else:
327
+ logical_dtype_code = 0
328
+ requant_block = max(128, triton.next_power_of_2(in_features))
329
+ _requantize_addmm_rows_kernel[(out_features,)](
330
+ qdata,
331
+ scale,
332
+ update,
333
+ in_features,
334
+ qdata.stride(0),
335
+ qdata.stride(1),
336
+ scale.stride(0),
337
+ update.stride(0),
338
+ update.stride(1),
339
+ beta,
340
+ alpha,
341
+ block_size=requant_block,
342
+ logical_dtype_code=logical_dtype_code,
343
+ has_base=beta != 0,
344
+ has_update=has_update,
345
+ num_warps=8,
346
+ )
347
+
348
+
349
+ @triton_convrot_int8_linear.register_fake
350
+ def _triton_convrot_int8_linear_fake(
351
+ activation: torch.Tensor,
352
+ weight: torch.Tensor,
353
+ _weight_scale: torch.Tensor,
354
+ _bias: torch.Tensor | None,
355
+ _group_size: int,
356
+ ) -> torch.Tensor:
357
+ return activation.new_empty((*activation.shape[:-1], weight.shape[0]))
@@ -0,0 +1,182 @@
1
+ """Internal backend selection for INT8 ConvRot operators."""
2
+
3
+ import math
4
+
5
+ import torch
6
+
7
+ from .reference import reference_addmm_, reference_linear, validate_storage
8
+
9
+ try:
10
+ from .backends.triton import (
11
+ triton_convrot_int8_addmm_ as _triton_addmm_,
12
+ )
13
+ from .backends.triton import (
14
+ triton_convrot_int8_linear as _triton_linear,
15
+ )
16
+ except ModuleNotFoundError as exc:
17
+ if exc.name != "triton":
18
+ raise
19
+ _triton_addmm_ = None
20
+ _triton_linear = None
21
+
22
+
23
+ def _can_use_triton(activation: torch.Tensor, qdata: torch.Tensor) -> bool:
24
+ return (
25
+ _triton_linear is not None
26
+ and activation.device.type == "cuda"
27
+ and activation.device == qdata.device
28
+ and activation.dtype in (torch.float16, torch.bfloat16, torch.float32)
29
+ and torch.cuda.get_device_capability(activation.device) >= (7, 5)
30
+ )
31
+
32
+
33
+ def _validate_scalar(value: int | float | complex, name: str) -> float:
34
+ if isinstance(value, complex):
35
+ raise TypeError(f"ConvRot INT8 addmm_ {name} must be a real number, got {value}")
36
+ converted = float(value)
37
+ if not math.isfinite(converted):
38
+ raise ValueError(f"ConvRot INT8 addmm_ {name} must be finite, got {value}")
39
+ return converted
40
+
41
+
42
+ def _validate_addmm(
43
+ qdata: torch.Tensor,
44
+ scale: torch.Tensor,
45
+ dtype: torch.dtype,
46
+ group_size: int,
47
+ mat1: torch.Tensor,
48
+ mat2: torch.Tensor,
49
+ ) -> None:
50
+ validate_storage(qdata, scale, group_size, dtype)
51
+ if qdata.device.type == "meta":
52
+ raise ValueError("ConvRot INT8 addmm_ cannot update a meta tensor without values")
53
+ if mat1.ndim != 2 or mat2.ndim != 2:
54
+ raise ValueError(
55
+ "ConvRot INT8 addmm_ matrices must be 2-D, "
56
+ f"got shapes {tuple(mat1.shape)} and {tuple(mat2.shape)}"
57
+ )
58
+ expected_mat1 = (qdata.shape[0], mat2.shape[0])
59
+ expected_mat2 = (mat1.shape[1], qdata.shape[1])
60
+ if tuple(mat1.shape) != expected_mat1 or tuple(mat2.shape) != expected_mat2:
61
+ raise ValueError(
62
+ "ConvRot INT8 addmm_ shape mismatch: expected "
63
+ f"mat1 {expected_mat1} and mat2 {expected_mat2} for weight {tuple(qdata.shape)}, "
64
+ f"got {tuple(mat1.shape)} and {tuple(mat2.shape)}"
65
+ )
66
+ if mat1.device != qdata.device or mat2.device != qdata.device:
67
+ raise ValueError(
68
+ "ConvRot INT8 addmm_ weight and matrices must share a device, "
69
+ f"got {qdata.device}/{mat1.device}/{mat2.device}"
70
+ )
71
+ if mat1.dtype is not dtype or mat2.dtype is not dtype:
72
+ raise ValueError(
73
+ "ConvRot INT8 addmm_ matrices must match the weight's logical dtype, "
74
+ f"got {dtype}/{mat1.dtype}/{mat2.dtype}"
75
+ )
76
+ if mat1.layout is not torch.strided or mat2.layout is not torch.strided:
77
+ raise ValueError("ConvRot INT8 addmm_ matrices must use strided layout")
78
+ if torch.is_grad_enabled() and (mat1.requires_grad or mat2.requires_grad):
79
+ raise RuntimeError(
80
+ "ConvRot INT8 addmm_ does not support autograd; detach the matrices or use no_grad"
81
+ )
82
+
83
+
84
+ def _can_use_triton_addmm(qdata: torch.Tensor, mat1: torch.Tensor) -> bool:
85
+ return (
86
+ _triton_addmm_ is not None
87
+ and qdata.device.type == "cuda"
88
+ and mat1.device == qdata.device
89
+ and torch.cuda.get_device_capability(qdata.device) >= (7, 5)
90
+ )
91
+
92
+
93
+ @torch.library.custom_op(
94
+ "piper_kernels::convrot_int8_addmm_",
95
+ mutates_args=("qdata", "scale"),
96
+ )
97
+ def _convrot_int8_addmm_op(
98
+ qdata: torch.Tensor,
99
+ scale: torch.Tensor,
100
+ mat1: torch.Tensor,
101
+ mat2: torch.Tensor,
102
+ group_size: int,
103
+ beta: float,
104
+ alpha: float,
105
+ ) -> None:
106
+ if _can_use_triton_addmm(qdata, mat1):
107
+ assert _triton_addmm_ is not None
108
+ _triton_addmm_(qdata, scale, mat1, mat2, group_size, beta, alpha)
109
+ return
110
+ reference_addmm_(qdata, scale, mat1, mat2, group_size, beta, alpha)
111
+
112
+
113
+ @_convrot_int8_addmm_op.register_fake
114
+ def _convrot_int8_addmm_fake(
115
+ _qdata: torch.Tensor,
116
+ _scale: torch.Tensor,
117
+ _mat1: torch.Tensor,
118
+ _mat2: torch.Tensor,
119
+ _group_size: int,
120
+ _beta: float,
121
+ _alpha: float,
122
+ ) -> None:
123
+ return None
124
+
125
+
126
+ def _addmm_(
127
+ qdata: torch.Tensor,
128
+ scale: torch.Tensor,
129
+ dtype: torch.dtype,
130
+ group_size: int,
131
+ mat1: torch.Tensor,
132
+ mat2: torch.Tensor,
133
+ *,
134
+ beta: int | float | complex = 1,
135
+ alpha: int | float | complex = 1,
136
+ ) -> None:
137
+ """Apply the logical ``beta * weight + alpha * mat1 @ mat2`` update in place."""
138
+ _validate_addmm(qdata, scale, dtype, group_size, mat1, mat2)
139
+ beta_float = _validate_scalar(beta, "beta")
140
+ alpha_float = _validate_scalar(alpha, "alpha")
141
+ if beta_float == 1 and alpha_float == 0:
142
+ return
143
+ _convrot_int8_addmm_op(
144
+ qdata,
145
+ scale,
146
+ mat1,
147
+ mat2,
148
+ group_size,
149
+ beta_float,
150
+ alpha_float,
151
+ )
152
+
153
+
154
+ def _linear(
155
+ activation: torch.Tensor,
156
+ qdata: torch.Tensor,
157
+ scale: torch.Tensor,
158
+ group_size: int,
159
+ bias: torch.Tensor | None = None,
160
+ ) -> torch.Tensor:
161
+ """Apply raw ConvRot INT8 storage to a floating-point activation.
162
+
163
+ ``qdata`` stores the two-dimensional INT8 weight in its rotated basis,
164
+ and ``scale`` contains one float32 value per output channel. This is the
165
+ internal storage-level ABI. Consumers should call
166
+ :func:`torch.nn.functional.linear` with a ``ConvRotInt8Tensor`` weight.
167
+ """
168
+ validate_storage(qdata, scale, group_size, activation.dtype)
169
+ if activation.ndim == 0 or activation.shape[-1] != qdata.shape[1]:
170
+ actual = 0 if activation.ndim == 0 else activation.shape[-1]
171
+ raise ValueError(f"ConvRot linear input has {actual} features, expected {qdata.shape[1]}")
172
+ if activation.device != qdata.device:
173
+ raise ValueError(
174
+ "ConvRot activation and weight must share a device, "
175
+ f"got {activation.device}/{qdata.device}"
176
+ )
177
+ if bias is not None and not isinstance(bias, torch.Tensor):
178
+ raise TypeError(f"ConvRot linear bias must be a tensor or None, got {type(bias).__name__}")
179
+ if _can_use_triton(activation, qdata):
180
+ assert _triton_linear is not None
181
+ return _triton_linear(activation, qdata, scale, bias, group_size)
182
+ return reference_linear(activation, qdata, scale, group_size, bias)
@@ -0,0 +1,132 @@
1
+ """Portable reference implementation for INT8 ConvRot linear layers."""
2
+
3
+ import torch
4
+
5
+ from .._rotation import rotate_groups, validate_group_size
6
+
7
+ _SUPPORTED_LOGICAL_DTYPES = (torch.float16, torch.bfloat16, torch.float32)
8
+
9
+
10
+ def quantize_weight(
11
+ weight: torch.Tensor,
12
+ group_size: int,
13
+ ) -> tuple[torch.Tensor, torch.Tensor]:
14
+ """Rotate and quantize a dense two-dimensional weight per output row."""
15
+ validate_group_size(group_size)
16
+ if weight.ndim != 2:
17
+ raise ValueError(
18
+ f"ConvRot INT8 high-precision weight must be 2-D, got shape {tuple(weight.shape)}"
19
+ )
20
+ if weight.dtype not in _SUPPORTED_LOGICAL_DTYPES:
21
+ raise ValueError(
22
+ "ConvRot INT8 high-precision weight must use float16, bfloat16, or float32, "
23
+ f"got {weight.dtype}"
24
+ )
25
+ if weight.device.type == "meta":
26
+ raise ValueError("ConvRot INT8 cannot quantize a meta tensor without values")
27
+ if weight.shape[1] % group_size:
28
+ raise ValueError(
29
+ f"ConvRot in_features {weight.shape[1]} is not divisible by group size {group_size}"
30
+ )
31
+ return dynamic_quantize_rows(rotate_groups(weight, group_size))
32
+
33
+
34
+ def validate_storage(
35
+ qdata: torch.Tensor,
36
+ scale: torch.Tensor,
37
+ group_size: int,
38
+ dtype: torch.dtype,
39
+ ) -> None:
40
+ """Validate INT8 ConvRot storage and its logical floating-point dtype."""
41
+ validate_group_size(group_size)
42
+ if qdata.dtype is not torch.int8 or qdata.ndim != 2:
43
+ raise ValueError(
44
+ f"ConvRot INT8 qdata must be a 2-D int8 tensor, got {qdata.dtype} {qdata.shape}"
45
+ )
46
+ if qdata.shape[1] % group_size:
47
+ raise ValueError(
48
+ f"ConvRot in_features {qdata.shape[1]} is not divisible by group size {group_size}"
49
+ )
50
+ expected_scale_shape = (qdata.shape[0], 1)
51
+ if scale.dtype is not torch.float32 or tuple(scale.shape) != expected_scale_shape:
52
+ raise ValueError(
53
+ f"ConvRot INT8 scale must be float32 with shape {expected_scale_shape}, "
54
+ f"got {scale.dtype} {tuple(scale.shape)} for qdata {tuple(qdata.shape)}"
55
+ )
56
+ if scale.device != qdata.device:
57
+ raise ValueError(
58
+ f"ConvRot INT8 qdata and scale must share a device, got {qdata.device}/{scale.device}"
59
+ )
60
+ if dtype not in _SUPPORTED_LOGICAL_DTYPES:
61
+ raise ValueError(
62
+ f"ConvRot logical dtype must be float16, bfloat16, or float32, got {dtype}"
63
+ )
64
+
65
+
66
+ def dynamic_quantize_rows(value: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
67
+ """Dynamically quantize each row to signed INT8 with a float32 scale."""
68
+ scale = (value.float().abs().amax(dim=-1, keepdim=True) / 127.0).clamp(min=1e-30)
69
+ logical_scale = scale.to(value.dtype)
70
+ if value.dtype is torch.float16:
71
+ scale_underflowed = logical_scale == 0
72
+ safe_logical_scale = torch.where(
73
+ scale_underflowed,
74
+ torch.ones_like(logical_scale),
75
+ logical_scale,
76
+ )
77
+ scaled = value / safe_logical_scale
78
+ scaled = torch.where(scale_underflowed, value.float() / scale, scaled.float())
79
+ else:
80
+ # The minimum scale remains representable in bfloat16 and float32.
81
+ scaled = value / logical_scale
82
+ qdata = scaled.round().clamp(-128, 127).to(torch.int8)
83
+ return qdata, scale
84
+
85
+
86
+ def reference_addmm_(
87
+ qdata: torch.Tensor,
88
+ scale: torch.Tensor,
89
+ mat1: torch.Tensor,
90
+ mat2: torch.Tensor,
91
+ group_size: int,
92
+ beta: float,
93
+ alpha: float,
94
+ ) -> None:
95
+ """Add a matrix product to a logical ConvRot weight and requantize it in place."""
96
+ if beta == 0:
97
+ rotated_weight = torch.zeros(qdata.shape, device=qdata.device, dtype=mat1.dtype)
98
+ else:
99
+ rotated_weight = qdata.to(mat1.dtype) * scale.to(mat1.dtype)
100
+ rotated_mat2 = rotate_groups(mat2, group_size)
101
+ merged = torch.addmm(rotated_weight, mat1, rotated_mat2, beta=beta, alpha=alpha)
102
+ merged_qdata, merged_scale = dynamic_quantize_rows(merged)
103
+ qdata.copy_(merged_qdata)
104
+ scale.copy_(merged_scale)
105
+
106
+
107
+ def reference_linear(
108
+ activation: torch.Tensor,
109
+ qdata: torch.Tensor,
110
+ scale: torch.Tensor,
111
+ group_size: int,
112
+ bias: torch.Tensor | None = None,
113
+ ) -> torch.Tensor:
114
+ """Run the portable PyTorch ConvRot W8A8 linear implementation."""
115
+ original_shape = activation.shape
116
+ activation_2d = activation.reshape(-1, original_shape[-1])
117
+ rotated = rotate_groups(activation_2d, group_size)
118
+ activation_qdata, activation_scale = dynamic_quantize_rows(rotated)
119
+ if activation.device.type == "cpu":
120
+ accumulated = activation_qdata.to(torch.int32) @ qdata.T.to(torch.int32)
121
+ else:
122
+ # Float32 represents each INT8 product exactly. Only very long reductions
123
+ # can round the integer sum, which is preferable to rejecting the shape.
124
+ accumulated = activation_qdata.float() @ qdata.T.float()
125
+ result = (
126
+ accumulated.to(torch.float32)
127
+ * activation_scale.to(torch.float32)
128
+ * scale.reshape(1, -1).to(torch.float32)
129
+ )
130
+ if bias is not None:
131
+ result += bias.to(torch.float32)
132
+ return result.to(activation.dtype).reshape(*original_shape[:-1], qdata.shape[0])
@@ -0,0 +1,150 @@
1
+ """Tensor subclass representing rotated INT8 W8A8 weights."""
2
+
3
+ from collections.abc import Callable
4
+ from typing import Any, ClassVar
5
+
6
+ import torch
7
+ from torchao.utils import TorchAOBaseTensor
8
+
9
+ from .._rotation import rotate_groups
10
+ from .dispatch import _addmm_, _linear
11
+ from .reference import quantize_weight, validate_storage
12
+
13
+
14
+ class ConvRotInt8Tensor(TorchAOBaseTensor):
15
+ """INT8 rotated weight with per-output scale and logical floating dtype."""
16
+
17
+ tensor_data_names: ClassVar[list[str]] = ["qdata", "scale"]
18
+ tensor_attribute_names: ClassVar[list[str]] = ["group_size", "dtype"]
19
+
20
+ def __new__(
21
+ cls,
22
+ qdata: torch.Tensor,
23
+ scale: torch.Tensor,
24
+ group_size: int,
25
+ dtype: torch.dtype = torch.bfloat16,
26
+ ) -> "ConvRotInt8Tensor":
27
+ validate_storage(qdata, scale, group_size, dtype)
28
+ return torch.Tensor._make_wrapper_subclass(
29
+ cls,
30
+ qdata.shape,
31
+ device=qdata.device,
32
+ dtype=dtype,
33
+ requires_grad=False,
34
+ )
35
+
36
+ def __init__(
37
+ self,
38
+ qdata: torch.Tensor,
39
+ scale: torch.Tensor,
40
+ group_size: int,
41
+ dtype: torch.dtype = torch.bfloat16,
42
+ ) -> None:
43
+ super().__init__()
44
+ if self.dtype is not dtype:
45
+ raise RuntimeError(f"ConvRot wrapper dtype mismatch: {self.dtype} != {dtype}")
46
+ self.qdata = qdata
47
+ self.scale = scale
48
+ self.group_size = group_size
49
+
50
+ @classmethod
51
+ def from_packed(
52
+ cls,
53
+ qdata: torch.Tensor,
54
+ scale: torch.Tensor,
55
+ *,
56
+ group_size: int,
57
+ dtype: torch.dtype = torch.bfloat16,
58
+ ) -> "ConvRotInt8Tensor":
59
+ """Normalize serialized INT8 ConvRot storage into its canonical representation."""
60
+ return cls(
61
+ qdata.contiguous(),
62
+ scale.reshape(-1, 1).contiguous(),
63
+ group_size,
64
+ dtype,
65
+ )
66
+
67
+ @classmethod
68
+ def from_hp(
69
+ cls,
70
+ hp_tensor: torch.Tensor,
71
+ *,
72
+ group_size: int,
73
+ ) -> "ConvRotInt8Tensor":
74
+ """Rotate and quantize a high-precision weight into ConvRot INT8 storage."""
75
+ source = hp_tensor.detach()
76
+ qdata, scale = quantize_weight(source, group_size)
77
+ return cls(qdata.contiguous(), scale.contiguous(), group_size, source.dtype)
78
+
79
+ def dequantize(self) -> torch.Tensor:
80
+ """Recover the logical weight in the unrotated basis."""
81
+ rotated = self.qdata.to(self.dtype) * self.scale.to(self.dtype)
82
+ return rotate_groups(rotated, self.group_size)
83
+
84
+ def _stable_hash_for_caching(self) -> str:
85
+ """Return a metadata fingerprint for AOTAutograd's cross-process cache."""
86
+ return repr(
87
+ (
88
+ type(self).__qualname__,
89
+ tuple(self.shape),
90
+ self.stride(),
91
+ str(self.device),
92
+ str(self.dtype),
93
+ self.group_size,
94
+ tuple(self.qdata.shape),
95
+ self.qdata.stride(),
96
+ tuple(self.scale.shape),
97
+ self.scale.stride(),
98
+ )
99
+ )
100
+
101
+
102
+ @ConvRotInt8Tensor.implements(torch.ops.aten.linear.default)
103
+ @ConvRotInt8Tensor.implements_torch_function(torch.nn.functional.linear)
104
+ def _convrot_linear_dispatch(
105
+ _func: Callable[..., torch.Tensor],
106
+ _types: tuple[type, ...],
107
+ args: tuple[Any, ...],
108
+ _kwargs: dict[str, Any],
109
+ ) -> torch.Tensor:
110
+ activation = args[0]
111
+ weight = args[1]
112
+ bias = args[2] if len(args) > 2 else None
113
+ if not isinstance(activation, torch.Tensor) or not isinstance(weight, ConvRotInt8Tensor):
114
+ raise TypeError(
115
+ "ConvRot linear dispatch requires a tensor input and ConvRotInt8Tensor weight"
116
+ )
117
+ if bias is not None and not isinstance(bias, torch.Tensor):
118
+ raise TypeError(f"ConvRot linear bias must be a tensor or None, got {type(bias).__name__}")
119
+ return _linear(
120
+ activation,
121
+ weight.qdata,
122
+ weight.scale,
123
+ weight.group_size,
124
+ bias,
125
+ )
126
+
127
+
128
+ @ConvRotInt8Tensor.implements(torch.ops.aten.addmm_.default)
129
+ def _convrot_addmm_dispatch(
130
+ _func: Callable[..., torch.Tensor],
131
+ _types: tuple[type, ...],
132
+ args: tuple[Any, ...],
133
+ kwargs: dict[str, Any],
134
+ ) -> ConvRotInt8Tensor:
135
+ weight, mat1, mat2 = args
136
+ if not isinstance(weight, ConvRotInt8Tensor):
137
+ raise TypeError(f"ConvRot addmm_ weight must be ConvRotInt8Tensor, got {type(weight)}")
138
+ if not isinstance(mat1, torch.Tensor) or not isinstance(mat2, torch.Tensor):
139
+ raise TypeError("ConvRot addmm_ matrices must be tensors")
140
+ _addmm_(
141
+ weight.qdata,
142
+ weight.scale,
143
+ weight.dtype,
144
+ weight.group_size,
145
+ mat1,
146
+ mat2,
147
+ beta=kwargs.get("beta", 1),
148
+ alpha=kwargs.get("alpha", 1),
149
+ )
150
+ return weight
@@ -0,0 +1,50 @@
1
+ """Portable rotation primitives shared by ConvRot storage formats."""
2
+
3
+ import math
4
+ from functools import cache
5
+
6
+ import torch
7
+
8
+ SUPPORTED_GROUP_SIZES = (16, 64, 256)
9
+
10
+
11
+ def validate_group_size(group_size: int) -> None:
12
+ """Validate a regular block-Hadamard group size supported by ConvRot."""
13
+ if group_size not in SUPPORTED_GROUP_SIZES:
14
+ supported = ", ".join(map(str, SUPPORTED_GROUP_SIZES))
15
+ raise ValueError(f"ConvRot group size must be one of {supported}, got {group_size}")
16
+
17
+
18
+ @cache
19
+ def build_hadamard(
20
+ size: int,
21
+ device: torch.device | None = None,
22
+ dtype: torch.dtype = torch.float32,
23
+ ) -> torch.Tensor:
24
+ """Build ConvRot's normalized regular Hadamard matrix in its fixed H4 order."""
25
+ validate_group_size(size)
26
+ device = torch.device("cpu") if device is None else device
27
+ h4 = torch.tensor(
28
+ ((1, 1, 1, -1), (1, 1, -1, 1), (1, -1, 1, 1), (-1, 1, 1, 1)),
29
+ device=device,
30
+ dtype=dtype,
31
+ )
32
+ result = h4
33
+ current_size = 4
34
+ while current_size < size:
35
+ result = torch.kron(result, h4)
36
+ current_size *= 4
37
+ return result / math.sqrt(size)
38
+
39
+
40
+ def rotate_groups(value: torch.Tensor, group_size: int) -> torch.Tensor:
41
+ """Multiply groups along the final dimension by ConvRot's regular Hadamard."""
42
+ validate_group_size(group_size)
43
+ features = value.shape[-1]
44
+ if features % group_size:
45
+ raise ValueError(
46
+ f"ConvRot feature dimension {features} is not divisible by group size {group_size}"
47
+ )
48
+ matrix = build_hadamard(group_size, value.device, value.dtype)
49
+ grouped = value.reshape(-1, features // group_size, group_size)
50
+ return torch.matmul(grouped, matrix).reshape(value.shape)
@@ -0,0 +1,96 @@
1
+ Metadata-Version: 2.4
2
+ Name: piper-kernels
3
+ Version: 0.1.0
4
+ Summary: Reusable PyTorch inference operators and Triton kernels.
5
+ Keywords: inference,pytorch,quantization,triton
6
+ License-Expression: Apache-2.0
7
+ License-File: LICENSE
8
+ License-File: NOTICE
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Programming Language :: Python :: 3 :: Only
12
+ Classifier: Programming Language :: Python :: 3.14
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Requires-Dist: torch>=2.12
15
+ Requires-Dist: torchao>=0.17 ; extra == 'convrot'
16
+ Requires-Dist: triton>=3.7 ; sys_platform == 'linux' and extra == 'triton'
17
+ Requires-Python: >=3.14
18
+ Project-URL: Homepage, https://github.com/Boffee/piper-kernels
19
+ Project-URL: Repository, https://github.com/Boffee/piper-kernels.git
20
+ Project-URL: Issues, https://github.com/Boffee/piper-kernels/issues
21
+ Project-URL: Changelog, https://github.com/Boffee/piper-kernels/blob/main/CHANGELOG.md
22
+ Provides-Extra: convrot
23
+ Provides-Extra: triton
24
+ Description-Content-Type: text/markdown
25
+
26
+ # Piper Kernels
27
+
28
+ Reusable PyTorch inference operators and optimized kernels for the Piper ecosystem and
29
+ other consumers.
30
+
31
+ Piper Kernels requires Python 3.14 or newer.
32
+
33
+ The package owns operator semantics, portable PyTorch references, tensor subclasses,
34
+ and optimized backends. It deliberately does not know about model repositories,
35
+ checkpoint metadata, pipeline frameworks, or device-offloading policy.
36
+
37
+ ## Planned operators
38
+
39
+ | Package | Role |
40
+ |---|---|
41
+ | `piper_kernels.convrot` | ConvRot quantized tensors and linear operators; INT8 today, INT4 planned |
42
+ | `piper_kernels.attention` | Attention operators, including a future SageAttention backend |
43
+
44
+ ## ConvRot INT8
45
+
46
+ Quantize a dense weight, or wrap existing checkpoint storage without dequantizing it,
47
+ then use the resulting tensor as a normal linear weight:
48
+
49
+ ```python
50
+ import torch
51
+
52
+ from piper_kernels.convrot import ConvRotInt8Tensor
53
+
54
+ weight = ConvRotInt8Tensor.from_hp(dense_weight, group_size=64)
55
+ checkpoint_weight = ConvRotInt8Tensor.from_packed(qdata, scale, group_size=64)
56
+ output = torch.nn.functional.linear(activation, weight, bias)
57
+
58
+ # In-place low-rank update with the standard Tensor.addmm_ contract.
59
+ weight.addmm_(lora_b, lora_a, alpha=lora_strength)
60
+ ```
61
+
62
+ `addmm_` computes `weight = beta * weight + alpha * (mat1 @ mat2)` and requantizes
63
+ the result. It preserves the ConvRot tensor and packed storage identities, allowing
64
+ offload integrations to keep their existing buffers. Repeated updates are lossy, so
65
+ reload a pristine base weight before changing or removing a previously merged adapter.
66
+ This is an inference operation and does not support autograd.
67
+
68
+ The operator selects its Triton implementation on supported CUDA devices and otherwise
69
+ uses the portable PyTorch reference. Install the tensor format and optimized backend with
70
+ `piper-kernels[convrot,triton]`. The base package does not require TorchAO or Triton, so
71
+ future attention-only consumers do not inherit quantization-specific dependencies.
72
+
73
+ ## Dependency direction
74
+
75
+ Applications such as Piper consume this package. Integrations such as torch-offload may
76
+ optionally recognize its tensor types, but `piper-kernels` does not depend on either
77
+ project.
78
+
79
+ ## Development
80
+
81
+ ```shell
82
+ uv sync --dev
83
+ uv run pytest
84
+ uv run ruff check .
85
+ uv run pyright
86
+ uv build
87
+ ```
88
+
89
+ GPU tests use the `gpu` pytest marker. The pre-commit test hook hides CUDA so commits run
90
+ the portable suite; run `uv run pytest` directly to exercise installed GPU backends.
91
+
92
+ ## Releases
93
+
94
+ Releases follow the compatibility and release policy in [VERSIONING.md](VERSIONING.md).
95
+ Distribution artifacts are built from version tags and published to PyPI by GitHub Actions
96
+ using Trusted Publishing; maintainers do not upload releases from local environments.
@@ -0,0 +1,16 @@
1
+ piper_kernels/__init__.py,sha256=ggMj1e9e0LuF9GUeZnp7PvQaNCGBpIlhxTlOiBwU7W4,66
2
+ piper_kernels/attention/__init__.py,sha256=2k-ybu6cMcnR6QLY5UvPnTYVm_V8ympLOb7nmHnwE5o,37
3
+ piper_kernels/attention/backends/__init__.py,sha256=P4wocoqF6iqnSEItmSTAQmn8hRsB3K-Bkctea7u7twg,59
4
+ piper_kernels/convrot/__init__.py,sha256=ub9XxG2p8GAJsrh5mShJ9dR7hgsBoJGHUsdO_cd1iyo,152
5
+ piper_kernels/convrot/_int8/__init__.py,sha256=uTtxVgR-L8z7FQVNM5EyH5NPiIssrX2pDCMZMKo7bgM,44
6
+ piper_kernels/convrot/_int8/backends/__init__.py,sha256=AEekqIVKK5ymxVwDgHc_Nxh21IiM1naoxlxQaLi8I1M,62
7
+ piper_kernels/convrot/_int8/backends/triton.py,sha256=TZ2sVBb-YjnFeGSBQwcQgGD_Q73LMjjsybni7Q0RFp0,11015
8
+ piper_kernels/convrot/_int8/dispatch.py,sha256=MBdrxW_HMLKiJ6mq0SGRxvNRQeFmb_jAeNpAMqPXerM,6312
9
+ piper_kernels/convrot/_int8/reference.py,sha256=GLBxFlD-vqEvKwZ1B78vN2g0DK_eC9OIlHPyCRX_4T0,5191
10
+ piper_kernels/convrot/_int8/tensor.py,sha256=NSEO2DLpHFjhpwjMKzbOYYCeYuijQ9_9nSyTuzkSKIA,4833
11
+ piper_kernels/convrot/_rotation.py,sha256=8NqN9H4kQBQfDJ4r0P4TBDSjRapY48F5s3h11Bfffbc,1714
12
+ piper_kernels-0.1.0.dist-info/licenses/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
13
+ piper_kernels-0.1.0.dist-info/licenses/NOTICE,sha256=7TOge4kGFKahwn-Pzxp3yw3Y64VbKHoovlnRioY2oHQ,129
14
+ piper_kernels-0.1.0.dist-info/WHEEL,sha256=lrO5MD1WVAWzcbNy_L2BwtfPrcM3KfUFpbKYXLkJX4A,80
15
+ piper_kernels-0.1.0.dist-info/METADATA,sha256=5lC8vjSgc8SHHTxEs67S6fcJAnxw32AI8a8-kekQfs8,3800
16
+ piper_kernels-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: uv 0.12.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,201 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
@@ -0,0 +1,4 @@
1
+ Piper Kernels
2
+ Copyright 2026 Piper Kernels contributors
3
+
4
+ This product includes software developed by Piper Kernels contributors.