PyPI - compressed-tensors - Versions diffs - 0.4.0__py3-none-any.whl → 0.6.0__py3-none-any.whl - Mend

compressed-tensors 0.4.0py3-none-any.whl → 0.6.0py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (44) hide show

compressed_tensors/base.py +1 -0
compressed_tensors/compressors/__init__.py +5 -1
compressed_tensors/compressors/base.py +200 -8
compressed_tensors/compressors/dense.py +1 -1
compressed_tensors/compressors/marlin_24.py +11 -10
compressed_tensors/compressors/model_compressor.py +101 -13
compressed_tensors/compressors/naive_quantized.py +140 -0
compressed_tensors/compressors/pack_quantized.py +128 -132
compressed_tensors/compressors/sparse_bitmask.py +1 -1
compressed_tensors/config/base.py +8 -1
compressed_tensors/{compressors/utils → linear}/__init__.py +0 -6
compressed_tensors/linear/compressed_linear.py +87 -0
compressed_tensors/quantization/lifecycle/__init__.py +1 -0
compressed_tensors/quantization/lifecycle/apply.py +204 -44
compressed_tensors/quantization/lifecycle/calibration.py +22 -2
compressed_tensors/quantization/lifecycle/compressed.py +3 -1
compressed_tensors/quantization/lifecycle/forward.py +139 -61
compressed_tensors/quantization/lifecycle/helpers.py +80 -0
compressed_tensors/quantization/lifecycle/initialize.py +77 -13
compressed_tensors/quantization/observers/__init__.py +1 -0
compressed_tensors/quantization/observers/base.py +93 -14
compressed_tensors/quantization/observers/helpers.py +64 -11
compressed_tensors/quantization/observers/min_max.py +8 -0
compressed_tensors/quantization/observers/mse.py +162 -0
compressed_tensors/quantization/quant_args.py +139 -23
compressed_tensors/quantization/quant_config.py +35 -2
compressed_tensors/quantization/quant_scheme.py +112 -13
compressed_tensors/quantization/utils/helpers.py +68 -2
compressed_tensors/utils/__init__.py +5 -0
compressed_tensors/utils/helpers.py +44 -2
compressed_tensors/utils/offload.py +116 -0
compressed_tensors/utils/permute.py +70 -0
compressed_tensors/utils/safetensors_load.py +2 -0
compressed_tensors/{compressors/utils → utils}/semi_structured_conversions.py +1 -0
compressed_tensors/version.py +1 -1
{compressed_tensors-0.4.0.dist-info → compressed_tensors-0.6.0.dist-info}/METADATA +35 -22
compressed_tensors-0.6.0.dist-info/RECORD +52 -0
{compressed_tensors-0.4.0.dist-info → compressed_tensors-0.6.0.dist-info}/WHEEL +1 -1
compressed_tensors/compressors/int_quantized.py +0 -126
compressed_tensors/compressors/utils/helpers.py +0 -43
compressed_tensors-0.4.0.dist-info/RECORD +0 -48
/compressed_tensors/{compressors/utils → utils}/permutations_24.py +0 -0
{compressed_tensors-0.4.0.dist-info → compressed_tensors-0.6.0.dist-info}/LICENSE +0 -0
{compressed_tensors-0.4.0.dist-info → compressed_tensors-0.6.0.dist-info}/top_level.txt +0 -0

compressed_tensors/quantization/lifecycle/forward.py CHANGED Viewed

@@ -17,12 +17,15 @@ from math import ceil
 from typing import Optional
 import torch
+from compressed_tensors.quantization.observers.helpers import calculate_range
 from compressed_tensors.quantization.quant_args import (
     QuantizationArgs,
     QuantizationStrategy,
+    round_to_quantized_type,
 )
 from compressed_tensors.quantization.quant_config import QuantizationStatus
 from compressed_tensors.quantization.quant_scheme import QuantizationScheme
+from compressed_tensors.utils import safe_permute, update_parameter_data
 from torch.nn import Module
@@ -42,6 +45,7 @@ def quantize(
     zero_point: torch.Tensor,
     args: QuantizationArgs,
     dtype: Optional[torch.dtype] = None,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Quantize the input tensor x using the QuantizationStrategy specified in args.
@@ -55,16 +59,9 @@ def quantize(
     :param zero_point: zero point tensor
     :param args: quantization args dictating how to quantize x
     :param dtype: optional dtype to cast the quantized output to
+    :param g_idx: optional mapping from column index to group index
     :return: fake quantized tensor
     """
-    # ensure all tensors are on the same device
-    # assumes that the target device is the input
-    # tensor's device
-    if x.device != scale.device:
-        scale = scale.to(x.device)
-    if x.device != zero_point.device:
-        zero_point = zero_point.to(x.device)
     return _process_quantization(
         x=x,
         scale=scale,
@@ -73,6 +70,7 @@ def quantize(
         dtype=dtype,
         do_quantize=True,
         do_dequantize=False,
+        g_idx=g_idx,
     )
@@ -80,8 +78,10 @@ def quantize(
 def dequantize(
     x_q: torch.Tensor,
     scale: torch.Tensor,
-    zero_point: torch.Tensor,
+    zero_point: torch.Tensor = None,
     args: QuantizationArgs = None,
+    dtype: Optional[torch.dtype] = None,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Dequantize a quantized input tensor x_q based on the strategy specified in args. If
@@ -91,6 +91,8 @@ def dequantize(
     :param scale: scale tensor
     :param zero_point: zero point tensor
     :param args: quantization args used to quantize x_q
+    :param dtype: optional dtype to cast the dequantized output to
+    :param g_idx: optional mapping from column index to group index
     :return: dequantized float tensor
     """
     if args is None:
@@ -107,8 +109,12 @@ def dequantize(
         else:
             raise ValueError(
                 f"Could not infer a quantization strategy from scale with {scale.ndim} "
-                "dimmensions. Expected 0-2 dimmensions."
+                "dimmensions. Expected 0 or 2 dimmensions."
             )
+    if dtype is None:
+        dtype = scale.dtype
     return _process_quantization(
         x=x_q,
         scale=scale,
@@ -116,6 +122,8 @@ def dequantize(
         args=args,
         do_quantize=False,
         do_dequantize=True,
+        dtype=dtype,
+        g_idx=g_idx,
     )
@@ -125,6 +133,7 @@ def fake_quantize(
     scale: torch.Tensor,
     zero_point: torch.Tensor,
     args: QuantizationArgs,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Fake quantize the input tensor x by quantizing then dequantizing with
@@ -137,6 +146,7 @@ def fake_quantize(
     :param scale: scale tensor
     :param zero_point: zero point tensor
     :param args: quantization args dictating how to quantize x
+    :param g_idx: optional mapping from column index to group index
     :return: fake quantized tensor
     """
     return _process_quantization(
@@ -146,6 +156,7 @@ def fake_quantize(
         args=args,
         do_quantize=True,
         do_dequantize=True,
+        g_idx=g_idx,
     )
@@ -154,64 +165,85 @@ def _process_quantization(
     x: torch.Tensor,
     scale: torch.Tensor,
     zero_point: torch.Tensor,
+    g_idx: Optional[torch.Tensor],
     args: QuantizationArgs,
     dtype: Optional[torch.dtype] = None,
     do_quantize: bool = True,
     do_dequantize: bool = True,
 ) -> torch.Tensor:
-    bit_range = 2**args.num_bits
-    q_max = torch.tensor(bit_range / 2 - 1, device=x.device)
-    q_min = torch.tensor(-bit_range / 2, device=x.device)
+    q_min, q_max = calculate_range(args, x.device)
     group_size = args.group_size
     if args.strategy == QuantizationStrategy.GROUP:
-        if do_dequantize and not do_quantize:
-            # if dequantizing a quantized type infer the output type from the scale
-            output = torch.zeros_like(x, dtype=scale.dtype)
-        else:
-            output_dtype = dtype if dtype is not None else x.dtype
-            output = torch.zeros_like(x, dtype=output_dtype)
-        # TODO: vectorize the for loop
-        # TODO: fix genetric assumption about the tensor size for computing group
+        output_dtype = dtype if dtype is not None else x.dtype
+        output = torch.zeros_like(x).to(output_dtype)
+        columns = output.shape[1]
         # TODO: make validation step for inputs
         while scale.ndim < 2:
             # pad scale and zero point dims for slicing
             scale = scale.unsqueeze(1)
-            zero_point = zero_point.unsqueeze(1)
+            zero_point = zero_point.unsqueeze(1) if zero_point is not None else None
-        columns = x.shape[1]
         if columns >= group_size:
             if columns % group_size != 0:
                 raise ValueError(
-                    "tesnor column shape must be divisble "
+                    "tensor column shape must be divisble "
                     f"by the given group_size {group_size}"
                 )
-        for i in range(ceil(columns / group_size)):
-            # scale.shape should be [nchan, ndim]
-            # sc.shape should be [nchan, 1] after unsqueeze
-            sc = scale[:, i].view(-1, 1)
-            zp = zero_point[:, i].view(-1, 1)
-            idx = i * group_size
+        # support column-order (default) quantization as well as other orderings
+        # such as activation ordering. Below checks if g_idx has been initialized
+        is_column_order = g_idx is None or -1 in g_idx
+        if is_column_order:
+            num_groups = int(ceil(columns / group_size))
+            group_sizes = torch.full((num_groups,), group_size, dtype=torch.int)
+        else:
+            group_indices, group_sizes = torch.unique(g_idx, return_counts=True)
+            group_sizes = group_sizes[torch.argsort(group_indices)]
+            perm = torch.argsort(g_idx)
+            x = safe_permute(x, perm, dim=1)
+        # TODO: experiment with vectorizing for loop for performance
+        end = 0
+        for index, group_count in enumerate(group_sizes):
+            sc = scale[:, index].view(-1, 1)
+            zp = zero_point[:, index].view(-1, 1) if zero_point is not None else None
+            start = end
+            end = start + group_count
             if do_quantize:
-                output[:, idx : (idx + group_size)] = _quantize(
-                    x[:, idx : (idx + group_size)], sc, zp, q_min, q_max, dtype=dtype
+                output[:, start:end] = _quantize(
+                    x[:, start:end],
+                    sc,
+                    zp,
+                    q_min,
+                    q_max,
+                    args,
+                    dtype=dtype,
                 )
             if do_dequantize:
-                input = (
-                    output[:, idx : (idx + group_size)]
-                    if do_quantize
-                    else x[:, idx : (idx + group_size)]
-                )
-                output[:, idx : (idx + group_size)] = _dequantize(input, sc, zp)
+                input = output[:, start:end] if do_quantize else x[:, start:end]
+                output[:, start:end] = _dequantize(input, sc, zp)
+        if not is_column_order:
+            output = safe_permute(output, torch.argsort(perm), dim=1)
     else:  # covers channel, token and tensor strategies
         if do_quantize:
-            output = _quantize(x, scale, zero_point, q_min, q_max, dtype=dtype)
+            output = _quantize(
+                x,
+                scale,
+                zero_point,
+                q_min,
+                q_max,
+                args,
+                dtype=dtype,
+            )
         if do_dequantize:
             output = _dequantize(output if do_quantize else x, scale, zero_point)
@@ -228,7 +260,13 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
     @wraps(forward_func_orig)  # ensures docstring, names, etc are propagated
     def wrapped_forward(self, *args, **kwargs):
+        if not getattr(module, "quantization_enabled", True):
+            # quantization is disabled on forward passes, return baseline
+            # forward call
+            return forward_func_orig.__get__(module, module.__class__)(*args, **kwargs)
         input_ = args[0]
+        compressed = module.quantization_status == QuantizationStatus.COMPRESSED
         if scheme.input_activations is not None:
             # calibrate and (fake) quantize input activations when applicable
@@ -236,7 +274,7 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
                 module, input_, "input", scheme.input_activations
             )
-        if scheme.weights is not None:
+        if scheme.weights is not None and not compressed:
             # calibrate and (fake) quantize weights when applicable
             unquantized_weight = self.weight.data.clone()
             self.weight.data = maybe_calibrate_or_quantize(
@@ -255,7 +293,7 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
             )
         # restore back to unquantized_value
-        if scheme.weights is not None:
+        if scheme.weights is not None and not compressed:
             self.weight.data = unquantized_weight
         return output
@@ -269,33 +307,57 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
 def maybe_calibrate_or_quantize(
     module: Module, value: torch.Tensor, base_name: str, args: "QuantizationArgs"
 ) -> torch.Tensor:
-    # only run quantized for the included stages
-    if module.quantization_status not in {
-        QuantizationStatus.CALIBRATION,
-        QuantizationStatus.FROZEN,
-    }:
+    # don't run quantization if we haven't entered calibration mode
+    if module.quantization_status == QuantizationStatus.INITIALIZED:
         return value
+    # in compressed mode, the weight is already compressed and quantized so we don't
+    # need to run fake quantization
+    if (
+        module.quantization_status == QuantizationStatus.COMPRESSED
+        and base_name == "weight"
+    ):
+        return value
+    if value.numel() == 0:
+        # if the tensor is empty,
+        # skip quantization
+        return value
+    g_idx = getattr(module, "weight_g_idx", None)
     if args.dynamic:
         # dynamic quantization - get scale and zero point directly from observer
         observer = getattr(module, f"{base_name}_observer")
-        scale, zero_point = observer(value)
+        scale, zero_point = observer(value, g_idx=g_idx)
     else:
         # static quantization - get previous scale and zero point from layer
         scale = getattr(module, f"{base_name}_scale")
-        zero_point = getattr(module, f"{base_name}_zero_point")
+        zero_point = getattr(module, f"{base_name}_zero_point", None)
-        if module.quantization_status == QuantizationStatus.CALIBRATION:
+        if (
+            module.quantization_status == QuantizationStatus.CALIBRATION
+            and base_name != "weight"
+        ):
             # calibration mode - get new quant params from observer
             observer = getattr(module, f"{base_name}_observer")
-            updated_scale, updated_zero_point = observer(value)
+            updated_scale, updated_zero_point = observer(value, g_idx=g_idx)
             # update scale and zero point
-            device = next(module.parameters()).device
-            scale.data = updated_scale.to(device)
-            zero_point.data = updated_zero_point.to(device)
-    return fake_quantize(value, scale, zero_point, args)
+            update_parameter_data(module, updated_scale, f"{base_name}_scale")
+            update_parameter_data(module, updated_zero_point, f"{base_name}_zero_point")
+            scale = updated_scale
+            zero_point = updated_zero_point
+    return fake_quantize(
+        x=value,
+        scale=scale,
+        zero_point=zero_point,
+        args=args,
+        g_idx=g_idx,
+    )
 @torch.no_grad()
@@ -305,14 +367,20 @@ def _quantize(
     zero_point: torch.Tensor,
     q_min: torch.Tensor,
     q_max: torch.Tensor,
+    args: QuantizationArgs,
     dtype: Optional[torch.dtype] = None,
 ) -> torch.Tensor:
-    quantized_value = torch.clamp(
-        torch.round(x / scale + zero_point),
+    scaled = x / scale
+    if zero_point is not None:
+        scaled += zero_point.to(x.dtype)
+    # clamp first because cast isn't guaranteed to be saturated (ie for fp8)
+    clamped_value = torch.clamp(
+        scaled,
         q_min,
         q_max,
     )
+    quantized_value = round_to_quantized_type(clamped_value, args)
     if dtype is not None:
         quantized_value = quantized_value.to(dtype)
@@ -323,6 +391,16 @@ def _quantize(
 def _dequantize(
     x_q: torch.Tensor,
     scale: torch.Tensor,
-    zero_point: torch.Tensor,
+    zero_point: torch.Tensor = None,
+    dtype: Optional[torch.dtype] = None,
 ) -> torch.Tensor:
-    return (x_q - zero_point) * scale
+    dequant_value = x_q.to(scale.dtype)
+    if zero_point is not None:
+        dequant_value = dequant_value - zero_point.to(scale.dtype)
+    dequant_value = dequant_value * scale
+    if dtype is not None:
+        dequant_value = dequant_value.to(dtype)
+    return dequant_value

compressed_tensors/quantization/lifecycle/helpers.py ADDED Viewed

@@ -0,0 +1,80 @@
+# Copyright (c) 2021 - present / Neuralmagic, Inc. All Rights Reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#    http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Miscelaneous helpers for the quantization lifecycle
+"""
+from typing import Optional
+import torch
+from torch.nn import Module
+__all__ = [
+    "update_layer_weight_quant_params",
+    "enable_quantization",
+    "disable_quantization",
+]
+def update_layer_weight_quant_params(
+    layer: Module,
+    weight: Optional[torch.Tensor] = None,
+    g_idx: Optional[torch.Tensor] = None,
+    reset_obs: bool = False,
+):
+    """
+    Update quantization parameters on layer
+    :param layer: input layer
+    :param weight: weight to update quant params with, defaults to layer weight
+    :param g_idx: optional mapping from column index to group index
+    :param reset_obs: reset the observer before calculating quant params,
+        defaults to False
+    """
+    attached_weight = getattr(layer, "weight", None)
+    if weight is None:
+        weight = attached_weight
+    scale = getattr(layer, "weight_scale", None)
+    zero_point = getattr(layer, "weight_zero_point", None)
+    if g_idx is None:
+        g_idx = getattr(layer, "weight_g_idx", None)
+    observer = getattr(layer, "weight_observer", None)
+    if weight is None or observer is None or scale is None or zero_point is None:
+        # scale, zp, or observer not calibratable or weight not available
+        return
+    if reset_obs:
+        observer.reset()
+    if attached_weight is not None:
+        weight = weight.to(attached_weight.dtype)
+    updated_scale, updated_zero_point = observer(weight)
+    # update scale and zero point
+    device = next(layer.parameters()).device
+    scale.data = updated_scale.to(device)
+    zero_point.data = updated_zero_point.to(device)
+def enable_quantization(module: Module):
+    module.quantization_enabled = True
+def disable_quantization(module: Module):
+    module.quantization_enabled = False

compressed_tensors/quantization/lifecycle/initialize.py CHANGED Viewed

@@ -21,11 +21,13 @@ from compressed_tensors.quantization.lifecycle.forward import (
     wrap_module_forward_quantized,
 )
 from compressed_tensors.quantization.quant_args import (
+    ActivationOrdering,
     QuantizationArgs,
     QuantizationStrategy,
 )
 from compressed_tensors.quantization.quant_config import QuantizationStatus
 from compressed_tensors.quantization.quant_scheme import QuantizationScheme
+from compressed_tensors.utils import get_execution_device, is_module_offloaded
 from torch.nn import Module, Parameter
@@ -40,6 +42,7 @@ _LOGGER = logging.getLogger(__name__)
 def initialize_module_for_quantization(
     module: Module,
     scheme: Optional[QuantizationScheme] = None,
+    force_zero_point: bool = True,
 ):
     """
     attaches appropriate scales, zero points, and observers to a layer
@@ -51,6 +54,8 @@ def initialize_module_for_quantization(
     :param scheme: scheme to use for quantization. if None is provided,
         will attempt to use scheme stored in the module under `quantization_scheme`,
         if not provided, the layer will be skipped
+    :param force_zero_point: whether to force initialization of a zero point for
+        symmetric quantization
     """
     scheme = scheme or getattr(module, "quantization_scheme", None)
     if scheme is None:
@@ -58,14 +63,18 @@ def initialize_module_for_quantization(
         return
     if scheme.input_activations is not None:
-        _initialize_scale_zero_point_observer(module, "input", scheme.input_activations)
+        _initialize_scale_zero_point_observer(
+            module, "input", scheme.input_activations, force_zero_point=force_zero_point
+        )
     if scheme.weights is not None:
         if hasattr(module, "weight"):
-            weight_shape = None
-            if isinstance(module, torch.nn.Linear):
-                weight_shape = module.weight.shape
+            weight_shape = module.weight.shape
             _initialize_scale_zero_point_observer(
-                module, "weight", scheme.weights, weight_shape=weight_shape
+                module,
+                "weight",
+                scheme.weights,
+                weight_shape=weight_shape,
+                force_zero_point=force_zero_point,
             )
         else:
             _LOGGER.warning(
@@ -75,21 +84,58 @@ def initialize_module_for_quantization(
             )
     if scheme.output_activations is not None:
         _initialize_scale_zero_point_observer(
-            module, "output", scheme.output_activations
+            module,
+            "output",
+            scheme.output_activations,
+            force_zero_point=force_zero_point,
         )
     module.quantization_scheme = scheme
     module.quantization_status = QuantizationStatus.INITIALIZED
+    offloaded = False
+    if is_module_offloaded(module):
+        try:
+            from accelerate.hooks import add_hook_to_module, remove_hook_from_module
+            from accelerate.utils import PrefixedDataset
+        except ModuleNotFoundError:
+            raise ModuleNotFoundError(
+                "Offloaded model detected. To use CPU offloading with "
+                "compressed-tensors the `accelerate` package must be installed, "
+                "run `pip install compressed-tensors[accelerate]`"
+            )
+        offloaded = True
+        hook = module._hf_hook
+        prefix_dict = module._hf_hook.weights_map
+        new_prefix = {}
+        # recreate the prefix dict (since it is immutable)
+        # and add quantization parameters
+        for key, data in module.named_parameters():
+            if key not in prefix_dict:
+                new_prefix[f"{prefix_dict.prefix}{key}"] = data
+            else:
+                new_prefix[f"{prefix_dict.prefix}{key}"] = prefix_dict[key]
+        new_prefix_dict = PrefixedDataset(new_prefix, prefix_dict.prefix)
+        remove_hook_from_module(module)
     # wrap forward call of module to perform quantized actions based on calltime status
     wrap_module_forward_quantized(module, scheme)
+    if offloaded:
+        # we need to re-add the hook for offloading now that we've wrapped forward
+        add_hook_to_module(module, hook)
+        if prefix_dict is not None:
+            module._hf_hook.weights_map = new_prefix_dict
 def _initialize_scale_zero_point_observer(
     module: Module,
     base_name: str,
     quantization_args: QuantizationArgs,
     weight_shape: Optional[torch.Size] = None,
+    force_zero_point: bool = True,
 ):
     # initialize observer module and attach as submodule
     observer = quantization_args.get_observer()
@@ -99,6 +145,8 @@ def _initialize_scale_zero_point_observer(
         return  # no need to register a scale and zero point for a dynamic observer
     device = next(module.parameters()).device
+    if is_module_offloaded(module):
+        device = get_execution_device(module)
     # infer expected scale/zero point shape
     expected_shape = 1  # per tensor
@@ -113,15 +161,31 @@ def _initialize_scale_zero_point_observer(
                 weight_shape[1] // quantization_args.group_size,
             )
-    # initializes empty scale and zero point parameters for the module
+    scale_dtype = module.weight.dtype
+    if scale_dtype not in [torch.float16, torch.bfloat16, torch.float32]:
+        scale_dtype = torch.float16
+    # initializes empty scale, zero point, and g_idx parameters for the module
     init_scale = Parameter(
-        torch.empty(expected_shape, dtype=module.weight.dtype, device=device),
+        torch.empty(expected_shape, dtype=scale_dtype, device=device),
         requires_grad=False,
     )
     module.register_parameter(f"{base_name}_scale", init_scale)
-    init_zero_point = Parameter(
-        torch.empty(expected_shape, device=device, dtype=int),
-        requires_grad=False,
-    )
-    module.register_parameter(f"{base_name}_zero_point", init_zero_point)
+    if force_zero_point or not quantization_args.symmetric:
+        zp_dtype = quantization_args.pytorch_dtype()
+        init_zero_point = Parameter(
+            torch.zeros(expected_shape, device=device, dtype=zp_dtype),
+            requires_grad=False,
+        )
+        module.register_parameter(f"{base_name}_zero_point", init_zero_point)
+    # only grouped activation ordering has g_idx
+    if quantization_args.actorder == ActivationOrdering.GROUP:
+        g_idx_shape = (weight_shape[1],)
+        g_idx_dtype = torch.int
+        init_g_idx = Parameter(
+            torch.full(g_idx_shape, -1, device=device, dtype=g_idx_dtype),
+            requires_grad=False,
+        )
+        module.register_parameter(f"{base_name}_g_idx", init_g_idx)

compressed_tensors/quantization/observers/__init__.py CHANGED Viewed

@@ -19,3 +19,4 @@ from .helpers import *
 from .base import *
 from .memoryless import *
 from .min_max import *
+from .mse import *

compressed-tensors 0.4.0__py3-none-any.whl → 0.6.0__py3-none-any.whl

compressed-tensors 0.4.0py3-none-any.whl → 0.6.0py3-none-any.whl