PyPI - compressed-tensors - Versions diffs - 0.5.0__py3-none-any.whl → 0.7.0__py3-none-any.whl - Mend

compressed-tensors 0.5.0py3-none-any.whl → 0.7.0py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (51) hide show

compressed_tensors/quantization/lifecycle/compressed.py CHANGED Viewed

@@ -49,8 +49,9 @@ def compress_quantized_weights(module: Module):
     weight = getattr(module, "weight", None)
     scale = getattr(module, "weight_scale", None)
     zero_point = getattr(module, "weight_zero_point", None)
+    g_idx = getattr(module, "weight_g_idx", None)
-    if weight is None or scale is None or zero_point is None:
+    if weight is None or scale is None:
         # no weight, scale, or ZP, nothing to do
         # mark as compressed here to maintain consistent status throughout the model
@@ -62,6 +63,7 @@ def compress_quantized_weights(module: Module):
         x=weight,
         scale=scale,
         zero_point=zero_point,
+        g_idx=g_idx,
         args=scheme.weights,
         dtype=torch.int8,
     )

compressed_tensors/quantization/lifecycle/forward.py CHANGED Viewed

@@ -14,9 +14,10 @@
 from functools import wraps
 from math import ceil
-from typing import Optional
+from typing import Callable, Optional
 import torch
+from compressed_tensors.quantization.cache import QuantizedKVParameterCache
 from compressed_tensors.quantization.observers.helpers import calculate_range
 from compressed_tensors.quantization.quant_args import (
     QuantizationArgs,
@@ -25,7 +26,7 @@ from compressed_tensors.quantization.quant_args import (
 )
 from compressed_tensors.quantization.quant_config import QuantizationStatus
 from compressed_tensors.quantization.quant_scheme import QuantizationScheme
-from compressed_tensors.utils import update_parameter_data
+from compressed_tensors.utils import safe_permute, update_parameter_data
 from torch.nn import Module
@@ -45,6 +46,7 @@ def quantize(
     zero_point: torch.Tensor,
     args: QuantizationArgs,
     dtype: Optional[torch.dtype] = None,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Quantize the input tensor x using the QuantizationStrategy specified in args.
@@ -58,15 +60,9 @@ def quantize(
     :param zero_point: zero point tensor
     :param args: quantization args dictating how to quantize x
     :param dtype: optional dtype to cast the quantized output to
+    :param g_idx: optional mapping from column index to group index
     :return: fake quantized tensor
     """
-    # ensure all tensors are on the same device
-    # assumes that the target device is the input
-    # tensor's device
-    if x.device != scale.device:
-        scale = scale.to(x.device)
-    if x.device != zero_point.device:
-        zero_point = zero_point.to(x.device)
     return _process_quantization(
         x=x,
@@ -76,6 +72,7 @@ def quantize(
         dtype=dtype,
         do_quantize=True,
         do_dequantize=False,
+        g_idx=g_idx,
     )
@@ -86,6 +83,7 @@ def dequantize(
     zero_point: torch.Tensor = None,
     args: QuantizationArgs = None,
     dtype: Optional[torch.dtype] = None,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Dequantize a quantized input tensor x_q based on the strategy specified in args. If
@@ -96,6 +94,7 @@ def dequantize(
     :param zero_point: zero point tensor
     :param args: quantization args used to quantize x_q
     :param dtype: optional dtype to cast the dequantized output to
+    :param g_idx: optional mapping from column index to group index
     :return: dequantized float tensor
     """
     if args is None:
@@ -126,6 +125,7 @@ def dequantize(
         do_quantize=False,
         do_dequantize=True,
         dtype=dtype,
+        g_idx=g_idx,
     )
@@ -135,6 +135,7 @@ def fake_quantize(
     scale: torch.Tensor,
     zero_point: torch.Tensor,
     args: QuantizationArgs,
+    g_idx: Optional[torch.Tensor] = None,
 ) -> torch.Tensor:
     """
     Fake quantize the input tensor x by quantizing then dequantizing with
@@ -147,6 +148,7 @@ def fake_quantize(
     :param scale: scale tensor
     :param zero_point: zero point tensor
     :param args: quantization args dictating how to quantize x
+    :param g_idx: optional mapping from column index to group index
     :return: fake quantized tensor
     """
     return _process_quantization(
@@ -156,6 +158,7 @@ def fake_quantize(
         args=args,
         do_quantize=True,
         do_dequantize=True,
+        g_idx=g_idx,
     )
@@ -165,20 +168,18 @@ def _process_quantization(
     scale: torch.Tensor,
     zero_point: torch.Tensor,
     args: QuantizationArgs,
+    g_idx: Optional[torch.Tensor] = None,
     dtype: Optional[torch.dtype] = None,
     do_quantize: bool = True,
     do_dequantize: bool = True,
 ) -> torch.Tensor:
     q_min, q_max = calculate_range(args, x.device)
     group_size = args.group_size
     if args.strategy == QuantizationStrategy.GROUP:
         output_dtype = dtype if dtype is not None else x.dtype
         output = torch.zeros_like(x).to(output_dtype)
-        # TODO: vectorize the for loop
-        # TODO: fix genetric assumption about the tensor size for computing group
+        columns = output.shape[1]
         # TODO: make validation step for inputs
@@ -187,23 +188,38 @@ def _process_quantization(
             scale = scale.unsqueeze(1)
             zero_point = zero_point.unsqueeze(1) if zero_point is not None else None
-        columns = x.shape[1]
         if columns >= group_size:
             if columns % group_size != 0:
                 raise ValueError(
-                    "tesnor column shape must be divisble "
+                    "tensor column shape must be divisble "
                     f"by the given group_size {group_size}"
                 )
-        for i in range(ceil(columns / group_size)):
-            # scale.shape should be [nchan, ndim]
-            # sc.shape should be [nchan, 1] after unsqueeze
-            sc = scale[:, i].view(-1, 1)
-            zp = zero_point[:, i].view(-1, 1) if zero_point is not None else None
-            idx = i * group_size
+        # support column-order (default) quantization as well as other orderings
+        # such as activation ordering. Below checks if g_idx has been initialized
+        is_column_order = g_idx is None or -1 in g_idx
+        if is_column_order:
+            num_groups = int(ceil(columns / group_size))
+            group_sizes = torch.full((num_groups,), group_size, dtype=torch.int)
+        else:
+            group_indices, group_sizes = torch.unique(g_idx, return_counts=True)
+            group_sizes = group_sizes[torch.argsort(group_indices)]
+            perm = torch.argsort(g_idx)
+            x = safe_permute(x, perm, dim=1)
+        # TODO: experiment with vectorizing for loop for performance
+        end = 0
+        for index, group_count in enumerate(group_sizes):
+            sc = scale[:, index].view(-1, 1)
+            zp = zero_point[:, index].view(-1, 1) if zero_point is not None else None
+            start = end
+            end = start + group_count
             if do_quantize:
-                output[:, idx : (idx + group_size)] = _quantize(
-                    x[:, idx : (idx + group_size)],
+                output[:, start:end] = _quantize(
+                    x[:, start:end],
                     sc,
                     zp,
                     q_min,
@@ -211,13 +227,13 @@ def _process_quantization(
                     args,
                     dtype=dtype,
                 )
             if do_dequantize:
-                input = (
-                    output[:, idx : (idx + group_size)]
-                    if do_quantize
-                    else x[:, idx : (idx + group_size)]
-                )
-                output[:, idx : (idx + group_size)] = _dequantize(input, sc, zp)
+                input = output[:, start:end] if do_quantize else x[:, start:end]
+                output[:, start:end] = _dequantize(input, sc, zp)
+        if not is_column_order:
+            output = safe_permute(output, torch.argsort(perm), dim=1)
     else:  # covers channel, token and tensor strategies
         if do_quantize:
@@ -253,13 +269,15 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
         input_ = args[0]
+        compressed = module.quantization_status == QuantizationStatus.COMPRESSED
         if scheme.input_activations is not None:
             # calibrate and (fake) quantize input activations when applicable
             input_ = maybe_calibrate_or_quantize(
                 module, input_, "input", scheme.input_activations
             )
-        if scheme.weights is not None:
+        if scheme.weights is not None and not compressed:
             # calibrate and (fake) quantize weights when applicable
             unquantized_weight = self.weight.data.clone()
             self.weight.data = maybe_calibrate_or_quantize(
@@ -270,15 +288,17 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
         output = forward_func_orig.__get__(module, module.__class__)(
             input_, *args[1:], **kwargs
         )
         if scheme.output_activations is not None:
             # calibrate and (fake) quantize output activations when applicable
+            # kv_cache scales updated on model self_attn forward call in
+            # wrap_module_forward_quantized_attn
             output = maybe_calibrate_or_quantize(
                 module, output, "output", scheme.output_activations
             )
         # restore back to unquantized_value
-        if scheme.weights is not None:
+        if scheme.weights is not None and not compressed:
             self.weight.data = unquantized_weight
         return output
@@ -289,14 +309,63 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
     setattr(module, "forward", bound_wrapped_forward)
+def wrap_module_forward_quantized_attn(module: Module, scheme: QuantizationScheme):
+    # expects a module already initialized and injected with the parameters in
+    # initialize_module_for_quantization
+    if hasattr(module.forward, "__func__"):
+        forward_func_orig = module.forward.__func__
+    else:
+        forward_func_orig = module.forward.func
+    @wraps(forward_func_orig)  # ensures docstring, names, etc are propagated
+    def wrapped_forward(self, *args, **kwargs):
+        # kv cache stored under weights
+        if module.quantization_status == QuantizationStatus.CALIBRATION:
+            quantization_args: QuantizationArgs = scheme.output_activations
+            past_key_value: QuantizedKVParameterCache = quantization_args.get_kv_cache()
+            kwargs["past_key_value"] = past_key_value
+            # QuantizedKVParameterCache used for obtaining k_scale, v_scale only,
+            # does not store quantized_key_states and quantized_value_state
+            kwargs["use_cache"] = False
+            attn_forward: Callable = forward_func_orig.__get__(module, module.__class__)
+            past_key_value.reset_states()
+            rtn = attn_forward(*args, **kwargs)
+            update_parameter_data(
+                module, past_key_value.k_scales[module.layer_idx], "k_scale"
+            )
+            update_parameter_data(
+                module, past_key_value.v_scales[module.layer_idx], "v_scale"
+            )
+            return rtn
+        return forward_func_orig.__get__(module, module.__class__)(*args, **kwargs)
+    # bind wrapped forward to module class so reference to `self` is correct
+    bound_wrapped_forward = wrapped_forward.__get__(module, module.__class__)
+    # set forward to wrapped forward
+    setattr(module, "forward", bound_wrapped_forward)
 def maybe_calibrate_or_quantize(
     module: Module, value: torch.Tensor, base_name: str, args: "QuantizationArgs"
 ) -> torch.Tensor:
-    # only run quantized for the included stages
-    if module.quantization_status not in {
-        QuantizationStatus.CALIBRATION,
-        QuantizationStatus.FROZEN,
-    }:
+    # don't run quantization if we haven't entered calibration mode
+    if module.quantization_status == QuantizationStatus.INITIALIZED:
+        return value
+    # in compressed mode, the weight is already compressed and quantized so we don't
+    # need to run fake quantization
+    if (
+        module.quantization_status == QuantizationStatus.COMPRESSED
+        and base_name == "weight"
+    ):
         return value
     if value.numel() == 0:
@@ -304,14 +373,16 @@ def maybe_calibrate_or_quantize(
         # skip quantization
         return value
+    g_idx = getattr(module, "weight_g_idx", None)
     if args.dynamic:
         # dynamic quantization - get scale and zero point directly from observer
         observer = getattr(module, f"{base_name}_observer")
-        scale, zero_point = observer(value)
+        scale, zero_point = observer(value, g_idx=g_idx)
     else:
         # static quantization - get previous scale and zero point from layer
         scale = getattr(module, f"{base_name}_scale")
-        zero_point = getattr(module, f"{base_name}_zero_point")
+        zero_point = getattr(module, f"{base_name}_zero_point", None)
         if (
             module.quantization_status == QuantizationStatus.CALIBRATION
@@ -320,13 +391,22 @@ def maybe_calibrate_or_quantize(
             # calibration mode - get new quant params from observer
             observer = getattr(module, f"{base_name}_observer")
-            updated_scale, updated_zero_point = observer(value)
+            updated_scale, updated_zero_point = observer(value, g_idx=g_idx)
             # update scale and zero point
             update_parameter_data(module, updated_scale, f"{base_name}_scale")
             update_parameter_data(module, updated_zero_point, f"{base_name}_zero_point")
-    return fake_quantize(value, scale, zero_point, args)
+            scale = updated_scale
+            zero_point = updated_zero_point
+    return fake_quantize(
+        x=value,
+        scale=scale,
+        zero_point=zero_point,
+        args=args,
+        g_idx=g_idx,
+    )
 @torch.no_grad()
@@ -340,7 +420,9 @@ def _quantize(
     dtype: Optional[torch.dtype] = None,
 ) -> torch.Tensor:
-    scaled = x / scale + zero_point.to(x.dtype)
+    scaled = x / scale
+    if zero_point is not None:
+        scaled += zero_point.to(x.dtype)
     # clamp first because cast isn't guaranteed to be saturated (ie for fp8)
     clamped_value = torch.clamp(
         scaled,
@@ -361,11 +443,11 @@ def _dequantize(
     zero_point: torch.Tensor = None,
     dtype: Optional[torch.dtype] = None,
 ) -> torch.Tensor:
+    dequant_value = x_q.to(scale.dtype)
-    dequant_value = x_q
     if zero_point is not None:
         dequant_value = dequant_value - zero_point.to(scale.dtype)
-    dequant_value = dequant_value.to(scale.dtype) * scale
+    dequant_value = dequant_value * scale
     if dtype is not None:
         dequant_value = dequant_value.to(dtype)

compressed_tensors/quantization/lifecycle/frozen.py CHANGED Viewed

@@ -14,6 +14,7 @@
 from compressed_tensors.quantization.quant_config import QuantizationStatus
+from compressed_tensors.quantization.utils import is_kv_cache_quant_scheme
 from torch.nn import Module
@@ -44,7 +45,11 @@ def freeze_module_quantization(module: Module):
         delattr(module, "input_observer")
     if scheme.weights and not scheme.weights.dynamic:
         delattr(module, "weight_observer")
-    if scheme.output_activations and not scheme.output_activations.dynamic:
+    if (
+        scheme.output_activations
+        and not is_kv_cache_quant_scheme(scheme)
+        and not scheme.output_activations.dynamic
+    ):
         delattr(module, "output_observer")
     module.quantization_status = QuantizationStatus.FROZEN

compressed_tensors/quantization/lifecycle/helpers.py CHANGED Viewed

@@ -16,35 +16,15 @@
 Miscelaneous helpers for the quantization lifecycle
 """
 from torch.nn import Module
 __all__ = [
-    "update_layer_weight_quant_params",
     "enable_quantization",
     "disable_quantization",
 ]
-def update_layer_weight_quant_params(layer: Module):
-    weight = getattr(layer, "weight", None)
-    scale = getattr(layer, "weight_scale", None)
-    zero_point = getattr(layer, "weight_zero_point", None)
-    observer = getattr(layer, "weight_observer", None)
-    if weight is None or observer is None or scale is None or zero_point is None:
-        # scale, zp, or observer not calibratable or weight not available
-        return
-    updated_scale, updated_zero_point = observer(weight)
-    # update scale and zero point
-    device = next(layer.parameters()).device
-    scale.data = updated_scale.to(device)
-    zero_point.data = updated_zero_point.to(device)
 def enable_quantization(module: Module):
     module.quantization_enabled = True

compressed-tensors 0.5.0__py3-none-any.whl → 0.7.0__py3-none-any.whl

compressed-tensors 0.5.0py3-none-any.whl → 0.7.0py3-none-any.whl