PyPI - compressed-tensors-nightly - Versions diffs - 0.6.0.20240925__py3-none-any.whl → 0.6.0.20240928__py3-none-any.whl - Mend

compressed-tensors-nightly 0.6.0.20240925py3-none-any.whl → 0.6.0.20240928py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (16) hide show

compressed_tensors/base.py CHANGED Viewed

@@ -16,3 +16,4 @@ SPARSITY_CONFIG_NAME = "sparsity_config"
 QUANTIZATION_CONFIG_NAME = "quantization_config"
 COMPRESSION_CONFIG_NAME = "compression_config"
 KV_CACHE_SCHEME_NAME = "kv_cache_scheme"
+COMPRESSION_VERSION_NAME = "version"

compressed_tensors/compressors/model_compressor.py CHANGED Viewed

@@ -25,6 +25,7 @@ import transformers
 import compressed_tensors
 from compressed_tensors.base import (
     COMPRESSION_CONFIG_NAME,
+    COMPRESSION_VERSION_NAME,
     QUANTIZATION_CONFIG_NAME,
     SPARSITY_CONFIG_NAME,
 )
@@ -200,6 +201,7 @@ class ModelCompressor:
         # SparseAutoModel format
         quantization_config = deepcopy(compression_config)
         quantization_config.pop(SPARSITY_CONFIG_NAME, None)
+        quantization_config.pop(COMPRESSION_VERSION_NAME, None)
         if len(quantization_config) == 0:
             quantization_config = None
         return quantization_config
@@ -214,6 +216,11 @@ class ModelCompressor:
         self.sparsity_compressor = None
         self.quantization_compressor = None
+        if sparsity_config and sparsity_config.format == CompressionFormat.dense.value:
+            # ignore dense sparsity config
+            self.sparsity_config = None
         if sparsity_config is not None:
             self.sparsity_compressor = Compressor.load_from_registry(
                 sparsity_config.format, config=sparsity_config
@@ -252,62 +259,6 @@ class ModelCompressor:
                 compressed_state_dict
             )
-        # HACK (mgoin): Post-process step for kv cache scales to take the
-        # k/v_proj module `output_scale` parameters, and store them in the
-        # parent attention module as `k_scale` and `v_scale`
-        #
-        # Example:
-        #  Replace `model.layers.0.self_attn.k_proj.output_scale`
-        #  with    `model.layers.0.self_attn.k_scale`
-        if (
-            self.quantization_config is not None
-            and self.quantization_config.kv_cache_scheme is not None
-        ):
-            # HACK (mgoin): We assume the quantized modules in question
-            # will be k_proj and v_proj since those are the default targets.
-            # We check that both of these modules have output activation
-            # quantization, and additionally check that q_proj doesn't.
-            q_proj_has_no_quant_output = 0
-            k_proj_has_quant_output = 0
-            v_proj_has_quant_output = 0
-            for name, module in model.named_modules():
-                if not hasattr(module, "quantization_scheme"):
-                    # We still want to count non-quantized q_proj
-                    if name.endswith(".q_proj"):
-                        q_proj_has_no_quant_output += 1
-                    continue
-                out_act = module.quantization_scheme.output_activations
-                if name.endswith(".q_proj") and out_act is None:
-                    q_proj_has_no_quant_output += 1
-                elif name.endswith(".k_proj") and out_act is not None:
-                    k_proj_has_quant_output += 1
-                elif name.endswith(".v_proj") and out_act is not None:
-                    v_proj_has_quant_output += 1
-            assert (
-                q_proj_has_no_quant_output > 0
-                and k_proj_has_quant_output > 0
-                and v_proj_has_quant_output > 0
-            )
-            assert (
-                q_proj_has_no_quant_output
-                == k_proj_has_quant_output
-                == v_proj_has_quant_output
-            )
-            # Move all .k/v_proj.output_scale parameters to .k/v_scale
-            working_state_dict = {}
-            for key in compressed_state_dict.keys():
-                if key.endswith(".k_proj.output_scale"):
-                    new_key = key.replace(".k_proj.output_scale", ".k_scale")
-                    working_state_dict[new_key] = compressed_state_dict[key]
-                elif key.endswith(".v_proj.output_scale"):
-                    new_key = key.replace(".v_proj.output_scale", ".v_scale")
-                    working_state_dict[new_key] = compressed_state_dict[key]
-                else:
-                    working_state_dict[key] = compressed_state_dict[key]
-            compressed_state_dict = working_state_dict
         # HACK: Override the dtype_byte_size function in transformers to
         # support float8 types. Fix is posted upstream
         # https://github.com/huggingface/transformers/pull/30488
@@ -360,16 +311,18 @@ class ModelCompressor:
         with open(config_file_path, "r") as config_file:
             config_data = json.load(config_file)
-        config_data[COMPRESSION_CONFIG_NAME] = {}
+        config_data[QUANTIZATION_CONFIG_NAME] = {}
         if self.quantization_config is not None:
             quant_config_data = self.quantization_config.model_dump()
-            config_data[COMPRESSION_CONFIG_NAME] = quant_config_data
+            config_data[QUANTIZATION_CONFIG_NAME] = quant_config_data
         if self.sparsity_config is not None:
             sparsity_config_data = self.sparsity_config.model_dump()
-            config_data[COMPRESSION_CONFIG_NAME][
+            config_data[QUANTIZATION_CONFIG_NAME][
                 SPARSITY_CONFIG_NAME
             ] = sparsity_config_data
-        config_data[COMPRESSION_CONFIG_NAME]["version"] = compressed_tensors.__version__
+        config_data[QUANTIZATION_CONFIG_NAME][
+            COMPRESSION_VERSION_NAME
+        ] = compressed_tensors.__version__
         with open(config_file_path, "w") as config_file:
             json.dump(config_data, config_file, indent=2, sort_keys=True)

compressed_tensors/quantization/__init__.py CHANGED Viewed

@@ -19,3 +19,4 @@ from .quant_args import *
 from .quant_config import *
 from .quant_scheme import *
 from .lifecycle import *
+from .cache import QuantizedKVParameterCache

compressed_tensors/quantization/cache.py ADDED Viewed

@@ -0,0 +1,201 @@
+# Copyright (c) 2021 - present / Neuralmagic, Inc. All Rights Reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#    http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from enum import Enum
+from typing import Any, Dict, List, Optional, Tuple
+from compressed_tensors.quantization.observers import Observer
+from compressed_tensors.quantization.quant_args import QuantizationArgs
+from torch import Tensor
+from transformers import DynamicCache as HFDyanmicCache
+class KVCacheScaleType(Enum):
+    KEY = "k_scale"
+    VALUE = "v_scale"
+class QuantizedKVParameterCache(HFDyanmicCache):
+    """
+    Quantized KV cache used in the forward call based on HF's dynamic cache.
+    Quantization strategy (tensor, group, channel) set from Quantization arg's strategy
+    Singleton, so that the same cache gets reused in all forward call of self_attn.
+    Each time forward is called, .update() is called, and ._quantize(), ._dequantize()
+     gets called appropriately.
+    The size of tensor is
+     `[batch_size, num_heads, seq_len - residual_length, head_dim]`.
+    Triggered by adding kv_cache_scheme in the recipe.
+    Example:
+    ```python3
+    recipe = '''
+    quant_stage:
+        quant_modifiers:
+            QuantizationModifier:
+                kv_cache_scheme:
+                    num_bits: 8
+                    type: float
+                    strategy: tensor
+                    dynamic: false
+                    symmetric: true
+    '''
+    """
+    _instance = None
+    _initialized = False
+    def __new__(cls, *args, **kwargs):
+        """Singleton"""
+        if cls._instance is None:
+            cls._instance = super(QuantizedKVParameterCache, cls).__new__(cls)
+        return cls._instance
+    def __init__(self, quantization_args: QuantizationArgs):
+        if not self._initialized:
+            super().__init__()
+            self.quantization_args = quantization_args
+            self.k_observers: List[Observer] = []
+            self.v_observers: List[Observer] = []
+            # each index corresponds to layer_idx of the attention layer
+            self.k_scales: List[Tensor] = []
+            self.v_scales: List[Tensor] = []
+            self.k_zps: List[Tensor] = []
+            self.v_zps: List[Tensor] = []
+            self._initialized = True
+    def update(
+        self,
+        key_states: Tensor,
+        value_states: Tensor,
+        layer_idx: int,
+        cache_kwargs: Optional[Dict[str, Any]] = None,
+    ) -> Tuple[Tensor, Tensor]:
+        """
+        Get the k_scale and v_scale and output the
+         fakequant-ed key_states and value_states
+        """
+        if len(self.k_observers) <= layer_idx:
+            k_observer = self.quantization_args.get_observer()
+            v_observer = self.quantization_args.get_observer()
+            self.k_observers.append(k_observer)
+            self.v_observers.append(v_observer)
+        q_key_states = self._quantize(
+            key_states.contiguous(), KVCacheScaleType.KEY, layer_idx
+        )
+        q_value_states = self._quantize(
+            value_states.contiguous(), KVCacheScaleType.VALUE, layer_idx
+        )
+        qdq_key_states = self._dequantize(q_key_states, KVCacheScaleType.KEY, layer_idx)
+        qdq_value_states = self._dequantize(
+            q_value_states, KVCacheScaleType.VALUE, layer_idx
+        )
+        keys_to_return, values_to_return = qdq_key_states, qdq_value_states
+        return keys_to_return, values_to_return
+    def get_seq_length(self, layer_idx: Optional[int] = 0) -> int:
+        """
+        Returns the sequence length of the cached states.
+        A layer index can be optionally passed.
+        """
+        if len(self.key_cache) <= layer_idx:
+            return 0
+        # since we cannot get the seq_length of each layer directly and
+        # rely on `_seen_tokens` which is updated every "layer_idx" == 0,
+        # this is a hack to get the actual seq_length for the given layer_idx
+        # this part of code otherwise fails when used to
+        # verify attn_weight shape in some models
+        return self._seen_tokens if layer_idx == 0 else self._seen_tokens - 1
+    def reset_states(self):
+        """reset the kv states (used in calibration)"""
+        self.key_cache: List[Tensor] = []
+        self.value_cache: List[Tensor] = []
+        # Used in `generate` to keep tally of how many tokens the cache has seen
+        self._seen_tokens = 0
+        self._quantized_key_cache: List[Tensor] = []
+        self._quantized_value_cache: List[Tensor] = []
+    def reset(self):
+        """
+        Reset the instantiation, create new instance on init
+        """
+        QuantizedKVParameterCache._instance = None
+        QuantizedKVParameterCache._initialized = False
+    def _quantize(self, tensor, kv_type, layer_idx):
+        """Quantizes a key/value using a defined quantization method."""
+        from compressed_tensors.quantization.lifecycle.forward import quantize
+        if kv_type == KVCacheScaleType.KEY:  # key type
+            observer = self.k_observers[layer_idx]
+            scales = self.k_scales
+            zps = self.k_zps
+        else:
+            assert kv_type == KVCacheScaleType.VALUE
+            observer = self.v_observers[layer_idx]
+            scales = self.v_scales
+            zps = self.v_zps
+        scale, zp = observer(tensor)
+        if len(scales) <= layer_idx:
+            scales.append(scale)
+            zps.append(zp)
+        else:
+            scales[layer_idx] = scale
+            zps[layer_idx] = scale
+        q_tensor = quantize(
+            x=tensor,
+            scale=scale,
+            zero_point=zp,
+            args=self.quantization_args,
+        )
+        return q_tensor
+    def _dequantize(self, qtensor, kv_type, layer_idx):
+        """Dequantizes back the tensor that was quantized by `self._quantize()`"""
+        from compressed_tensors.quantization.lifecycle.forward import dequantize
+        if kv_type == KVCacheScaleType.KEY:
+            scale = self.k_scales[layer_idx]
+            zp = self.k_zps[layer_idx]
+        else:
+            assert kv_type == KVCacheScaleType.VALUE
+            scale = self.v_scales[layer_idx]
+            zp = self.v_zps[layer_idx]
+        qdq_tensor = dequantize(
+            x_q=qtensor,
+            scale=scale,
+            zero_point=zp,
+            args=self.quantization_args,
+        )
+        return qdq_tensor

compressed_tensors/quantization/lifecycle/apply.py CHANGED Viewed

@@ -43,6 +43,7 @@ from compressed_tensors.quantization.utils import (
     infer_quantization_status,
     is_kv_cache_quant_scheme,
     iter_named_leaf_modules,
+    iter_named_quantizable_modules,
 )
 from compressed_tensors.utils.helpers import fix_fsdp_module_name, replace_module
 from compressed_tensors.utils.offload import update_parameter_data
@@ -135,15 +136,23 @@ def apply_quantization_config(
     # list of submodules to ignore
     ignored_submodules = defaultdict(list)
     # mark appropriate layers for quantization by setting their quantization schemes
-    for name, submodule in iter_named_leaf_modules(model):
+    for name, submodule in iter_named_quantizable_modules(
+        model,
+        include_children=True,
+        include_attn=True,
+    ):  # child modules and attention modules
         # potentially fix module name to remove FSDP wrapper prefix
         name = fix_fsdp_module_name(name)
         if matches := find_name_or_class_matches(name, submodule, config.ignore):
             for match in matches:
                 ignored_submodules[match].append(name)
             continue  # layer matches ignore list, continue
         targets = find_name_or_class_matches(name, submodule, target_to_scheme)
         if targets:
+            # mark modules to be quantized by adding
+            # quant scheme to the matching layers
             scheme = _scheme_from_targets(target_to_scheme, targets, name)
             if run_compressed:
                 format = config.format
@@ -200,6 +209,9 @@ def process_kv_cache_config(
     :param config: the QuantizationConfig
     :return: the QuantizationConfig with additional "kv_cache" group
     """
+    if targets == KV_CACHE_TARGETS:
+        _LOGGER.info(f"KV cache targets set to default value of: {KV_CACHE_TARGETS}")
     kv_cache_dict = config.kv_cache_scheme.model_dump()
     kv_cache_scheme = QuantizationScheme(
         output_activations=QuantizationArgs(**kv_cache_dict),

compressed_tensors/quantization/lifecycle/forward.py CHANGED Viewed

@@ -14,9 +14,10 @@
 from functools import wraps
 from math import ceil
-from typing import Optional
+from typing import Callable, Optional
 import torch
+from compressed_tensors.quantization.cache import QuantizedKVParameterCache
 from compressed_tensors.quantization.observers.helpers import calculate_range
 from compressed_tensors.quantization.quant_args import (
     QuantizationArgs,
@@ -62,6 +63,7 @@ def quantize(
     :param g_idx: optional mapping from column index to group index
     :return: fake quantized tensor
     """
     return _process_quantization(
         x=x,
         scale=scale,
@@ -165,8 +167,8 @@ def _process_quantization(
     x: torch.Tensor,
     scale: torch.Tensor,
     zero_point: torch.Tensor,
-    g_idx: Optional[torch.Tensor],
     args: QuantizationArgs,
+    g_idx: Optional[torch.Tensor] = None,
     dtype: Optional[torch.dtype] = None,
     do_quantize: bool = True,
     do_dequantize: bool = True,
@@ -266,6 +268,7 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
             return forward_func_orig.__get__(module, module.__class__)(*args, **kwargs)
         input_ = args[0]
         compressed = module.quantization_status == QuantizationStatus.COMPRESSED
         if scheme.input_activations is not None:
@@ -285,9 +288,11 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
         output = forward_func_orig.__get__(module, module.__class__)(
             input_, *args[1:], **kwargs
         )
         if scheme.output_activations is not None:
             # calibrate and (fake) quantize output activations when applicable
+            # kv_cache scales updated on model self_attn forward call in
+            # wrap_module_forward_quantized_attn
             output = maybe_calibrate_or_quantize(
                 module, output, "output", scheme.output_activations
             )
@@ -304,6 +309,50 @@ def wrap_module_forward_quantized(module: Module, scheme: QuantizationScheme):
     setattr(module, "forward", bound_wrapped_forward)
+def wrap_module_forward_quantized_attn(module: Module, scheme: QuantizationScheme):
+    # expects a module already initialized and injected with the parameters in
+    # initialize_module_for_quantization
+    if hasattr(module.forward, "__func__"):
+        forward_func_orig = module.forward.__func__
+    else:
+        forward_func_orig = module.forward.func
+    @wraps(forward_func_orig)  # ensures docstring, names, etc are propagated
+    def wrapped_forward(self, *args, **kwargs):
+        # kv cache stored under weights
+        if module.quantization_status == QuantizationStatus.CALIBRATION:
+            quantization_args: QuantizationArgs = scheme.output_activations
+            past_key_value: QuantizedKVParameterCache = quantization_args.get_kv_cache()
+            kwargs["past_key_value"] = past_key_value
+            # QuantizedKVParameterCache used for obtaining k_scale, v_scale only,
+            # does not store quantized_key_states and quantized_value_state
+            kwargs["use_cache"] = False
+            attn_forward: Callable = forward_func_orig.__get__(module, module.__class__)
+            past_key_value.reset_states()
+            rtn = attn_forward(*args, **kwargs)
+            update_parameter_data(
+                module, past_key_value.k_scales[module.layer_idx], "k_scale"
+            )
+            update_parameter_data(
+                module, past_key_value.v_scales[module.layer_idx], "v_scale"
+            )
+            return rtn
+        return forward_func_orig.__get__(module, module.__class__)(*args, **kwargs)
+    # bind wrapped forward to module class so reference to `self` is correct
+    bound_wrapped_forward = wrapped_forward.__get__(module, module.__class__)
+    # set forward to wrapped forward
+    setattr(module, "forward", bound_wrapped_forward)
 def maybe_calibrate_or_quantize(
     module: Module, value: torch.Tensor, base_name: str, args: "QuantizationArgs"
 ) -> torch.Tensor:

compressed_tensors/quantization/lifecycle/frozen.py CHANGED Viewed

@@ -14,6 +14,7 @@
 from compressed_tensors.quantization.quant_config import QuantizationStatus
+from compressed_tensors.quantization.utils import is_kv_cache_quant_scheme
 from torch.nn import Module
@@ -44,7 +45,11 @@ def freeze_module_quantization(module: Module):
         delattr(module, "input_observer")
     if scheme.weights and not scheme.weights.dynamic:
         delattr(module, "weight_observer")
-    if scheme.output_activations and not scheme.output_activations.dynamic:
+    if (
+        scheme.output_activations
+        and not is_kv_cache_quant_scheme(scheme)
+        and not scheme.output_activations.dynamic
+    ):
         delattr(module, "output_observer")
     module.quantization_status = QuantizationStatus.FROZEN

compressed_tensors/quantization/lifecycle/initialize.py CHANGED Viewed

@@ -17,8 +17,10 @@ import logging
 from typing import Optional
 import torch
+from compressed_tensors.quantization.cache import KVCacheScaleType
 from compressed_tensors.quantization.lifecycle.forward import (
     wrap_module_forward_quantized,
+    wrap_module_forward_quantized_attn,
 )
 from compressed_tensors.quantization.quant_args import (
     ActivationOrdering,
@@ -27,6 +29,7 @@ from compressed_tensors.quantization.quant_args import (
 )
 from compressed_tensors.quantization.quant_config import QuantizationStatus
 from compressed_tensors.quantization.quant_scheme import QuantizationScheme
+from compressed_tensors.quantization.utils import is_kv_cache_quant_scheme
 from compressed_tensors.utils import get_execution_device, is_module_offloaded
 from torch.nn import Module, Parameter
@@ -62,72 +65,85 @@ def initialize_module_for_quantization(
         # no scheme passed and layer not targeted for quantization - skip
         return
-    if scheme.input_activations is not None:
-        _initialize_scale_zero_point_observer(
-            module, "input", scheme.input_activations, force_zero_point=force_zero_point
-        )
-    if scheme.weights is not None:
-        if hasattr(module, "weight"):
-            weight_shape = module.weight.shape
+    if is_attention_module(module):
+        # wrap forward call of module to perform
+        # quantized actions based on calltime status
+        wrap_module_forward_quantized_attn(module, scheme)
+        _initialize_attn_scales(module)
+    else:
+        if scheme.input_activations is not None:
             _initialize_scale_zero_point_observer(
                 module,
-                "weight",
-                scheme.weights,
-                weight_shape=weight_shape,
+                "input",
+                scheme.input_activations,
                 force_zero_point=force_zero_point,
             )
-        else:
-            _LOGGER.warning(
-                f"module type {type(module)} targeted for weight quantization but "
-                "has no attribute weight, skipping weight quantization "
-                f"for {type(module)}"
-            )
-    if scheme.output_activations is not None:
-        _initialize_scale_zero_point_observer(
-            module,
-            "output",
-            scheme.output_activations,
-            force_zero_point=force_zero_point,
-        )
-    module.quantization_scheme = scheme
-    module.quantization_status = QuantizationStatus.INITIALIZED
-    offloaded = False
-    if is_module_offloaded(module):
-        try:
-            from accelerate.hooks import add_hook_to_module, remove_hook_from_module
-            from accelerate.utils import PrefixedDataset
-        except ModuleNotFoundError:
-            raise ModuleNotFoundError(
-                "Offloaded model detected. To use CPU offloading with "
-                "compressed-tensors the `accelerate` package must be installed, "
-                "run `pip install compressed-tensors[accelerate]`"
-            )
-        offloaded = True
-        hook = module._hf_hook
-        prefix_dict = module._hf_hook.weights_map
-        new_prefix = {}
-        # recreate the prefix dict (since it is immutable)
-        # and add quantization parameters
-        for key, data in module.named_parameters():
-            if key not in prefix_dict:
-                new_prefix[f"{prefix_dict.prefix}{key}"] = data
+        if scheme.weights is not None:
+            if hasattr(module, "weight"):
+                weight_shape = None
+                if isinstance(module, torch.nn.Linear):
+                    weight_shape = module.weight.shape
+                _initialize_scale_zero_point_observer(
+                    module,
+                    "weight",
+                    scheme.weights,
+                    weight_shape=weight_shape,
+                    force_zero_point=force_zero_point,
+                )
             else:
-                new_prefix[f"{prefix_dict.prefix}{key}"] = prefix_dict[key]
-        new_prefix_dict = PrefixedDataset(new_prefix, prefix_dict.prefix)
-        remove_hook_from_module(module)
-    # wrap forward call of module to perform quantized actions based on calltime status
-    wrap_module_forward_quantized(module, scheme)
-    if offloaded:
-        # we need to re-add the hook for offloading now that we've wrapped forward
-        add_hook_to_module(module, hook)
-        if prefix_dict is not None:
-            module._hf_hook.weights_map = new_prefix_dict
+                _LOGGER.warning(
+                    f"module type {type(module)} targeted for weight quantization but "
+                    "has no attribute weight, skipping weight quantization "
+                    f"for {type(module)}"
+                )
+        if scheme.output_activations is not None:
+            if not is_kv_cache_quant_scheme(scheme):
+                _initialize_scale_zero_point_observer(
+                    module, "output", scheme.output_activations
+                )
+        module.quantization_scheme = scheme
+        module.quantization_status = QuantizationStatus.INITIALIZED
+        offloaded = False
+        if is_module_offloaded(module):
+            try:
+                from accelerate.hooks import add_hook_to_module, remove_hook_from_module
+                from accelerate.utils import PrefixedDataset
+            except ModuleNotFoundError:
+                raise ModuleNotFoundError(
+                    "Offloaded model detected. To use CPU offloading with "
+                    "compressed-tensors the `accelerate` package must be installed, "
+                    "run `pip install compressed-tensors[accelerate]`"
+                )
+            offloaded = True
+            hook = module._hf_hook
+            prefix_dict = module._hf_hook.weights_map
+            new_prefix = {}
+            # recreate the prefix dict (since it is immutable)
+            # and add quantization parameters
+            for key, data in module.named_parameters():
+                if key not in prefix_dict:
+                    new_prefix[f"{prefix_dict.prefix}{key}"] = data
+                else:
+                    new_prefix[f"{prefix_dict.prefix}{key}"] = prefix_dict[key]
+            new_prefix_dict = PrefixedDataset(new_prefix, prefix_dict.prefix)
+            remove_hook_from_module(module)
+        # wrap forward call of module to perform
+        # quantized actions based on calltime status
+        wrap_module_forward_quantized(module, scheme)
+        if offloaded:
+            # we need to re-add the hook for offloading now that we've wrapped forward
+            add_hook_to_module(module, hook)
+            if prefix_dict is not None:
+                module._hf_hook.weights_map = new_prefix_dict
 def _initialize_scale_zero_point_observer(
@@ -189,3 +205,34 @@ def _initialize_scale_zero_point_observer(
             requires_grad=False,
         )
         module.register_parameter(f"{base_name}_g_idx", init_g_idx)
+def is_attention_module(module: Module):
+    return "attention" in module.__class__.__name__.lower() and (
+        hasattr(module, "k_proj")
+        or hasattr(module, "v_proj")
+        or hasattr(module, "qkv_proj")
+    )
+def _initialize_attn_scales(module: Module) -> None:
+    """Initlaize k_scale, v_scale for  self_attn"""
+    expected_shape = 1  # per tensor
+    param = next(module.parameters())
+    scale_dtype = param.dtype
+    device = param.device
+    init_scale = Parameter(
+        torch.empty(expected_shape, dtype=scale_dtype, device=device),
+        requires_grad=False,
+    )
+    module.register_parameter(KVCacheScaleType.KEY.value, init_scale)
+    init_scale = Parameter(
+        torch.empty(expected_shape, dtype=scale_dtype, device=device),
+        requires_grad=False,
+    )
+    module.register_parameter(KVCacheScaleType.VALUE.value, init_scale)

compressed_tensors/quantization/quant_args.py CHANGED Viewed

@@ -122,6 +122,12 @@ class QuantizationArgs(BaseModel, use_enum_values=True):
         return Observer.load_from_registry(self.observer, quantization_args=self)
+    def get_kv_cache(self):
+        """Get the singleton KV Cache"""
+        from compressed_tensors.quantization.cache import QuantizedKVParameterCache
+        return QuantizedKVParameterCache(self)
     @field_validator("type", mode="before")
     def validate_type(cls, value) -> QuantizationType:
         if isinstance(value, str):

compressed_tensors/quantization/quant_config.py CHANGED Viewed

@@ -24,7 +24,7 @@ from compressed_tensors.quantization.quant_scheme import (
 from compressed_tensors.quantization.utils import (
     calculate_compression_ratio,
     is_module_quantized,
-    iter_named_leaf_modules,
+    iter_named_quantizable_modules,
     module_type,
     parse_out_kv_cache_args,
 )
@@ -177,7 +177,9 @@ class QuantizationConfig(BaseModel):
         quantization_status = None
         ignore = {}
         quantization_type_names = set()
-        for name, submodule in iter_named_leaf_modules(model):
+        for name, submodule in iter_named_quantizable_modules(
+            model, include_children=True, include_attn=True
+        ):
             layer_type = module_type(submodule)
             if not is_module_quantized(submodule):
                 if layer_type not in ignore:
@@ -241,6 +243,9 @@ class QuantizationConfig(BaseModel):
         )
     def requires_calibration_data(self):
+        if self.kv_cache_scheme is not None:
+            return True
         for _, scheme in self.config_groups.items():
             if scheme.input_activations is not None:
                 if not scheme.input_activations.dynamic:

compressed_tensors/quantization/utils/helpers.py CHANGED Viewed

@@ -13,8 +13,7 @@
 # limitations under the License.
 import logging
-import re
-from typing import List, Optional, Tuple
+from typing import Generator, List, Optional, Tuple
 import torch
 from compressed_tensors.quantization.observers.base import Observer
@@ -28,7 +27,6 @@ __all__ = [
     "infer_quantization_status",
     "is_module_quantized",
     "is_model_quantized",
-    "iter_named_leaf_modules",
     "module_type",
     "calculate_compression_ratio",
     "get_torch_bit_depth",
@@ -36,9 +34,14 @@ __all__ = [
     "parse_out_kv_cache_args",
     "KV_CACHE_TARGETS",
     "is_kv_cache_quant_scheme",
+    "iter_named_leaf_modules",
+    "iter_named_quantizable_modules",
 ]
-KV_CACHE_TARGETS = ["re:.*k_proj", "re:.*v_proj"]
+# target the self_attn layer
+# QuantizedKVParameterCache is responsible for obtaining the k_scale and v_scale
+KV_CACHE_TARGETS = ["re:.*self_attn$"]
 _LOGGER: logging.Logger = logging.getLogger(__name__)
@@ -106,11 +109,10 @@ def module_type(module: Module) -> str:
     return type(module).__name__
-def iter_named_leaf_modules(model: Module) -> Tuple[str, Module]:
+def iter_named_leaf_modules(model: Module) -> Generator[Tuple[str, Module], None, None]:
     """
     Yields modules that do not have any submodules except observers. The observers
     themselves are not yielded
     :param model: model to get leaf modules of
     :returns: generator tuple of (name, leaf_submodule)
     """
@@ -128,6 +130,37 @@ def iter_named_leaf_modules(model: Module) -> Tuple[str, Module]:
                 yield name, submodule
+def iter_named_quantizable_modules(
+    model: Module, include_children: bool = True, include_attn: bool = False
+) -> Generator[Tuple[str, Module], None, None]:
+    """
+    Yield name and submodule of
+    - leaf modules, set by include_children
+    - attention modyles, set by include_attn
+    :param model: model to get leaf modules of
+    :param include_children: flag to get the leaf modules
+    :param inlcude_attn: flag to get the attention modules
+    :returns: generator tuple of (name, submodule)
+    """
+    for name, submodule in model.named_modules():
+        if include_children:
+            children = list(submodule.children())
+            if len(children) == 0 and not isinstance(submodule, Observer):
+                yield name, submodule
+            else:
+                has_non_observer_children = False
+                for child in children:
+                    if not isinstance(child, Observer):
+                        has_non_observer_children = True
+                if not has_non_observer_children:
+                    yield name, submodule
+        if include_attn:
+            if name.endswith("self_attn"):
+                yield name, submodule
 def get_torch_bit_depth(value: torch.Tensor) -> int:
     """
     Determine the number of bits used to represent the dtype of a tensor
@@ -204,19 +237,11 @@ def is_kv_cache_quant_scheme(scheme: QuantizationScheme) -> bool:
     :param scheme: The QuantizationScheme to investigate
     :return: boolean flag
     """
-    if len(scheme.targets) == 1:
-        # match on the KV_CACHE_TARGETS regex pattern
-        # if there is only one target
-        is_match_targets = any(
-            [re.match(pattern[3:], scheme.targets[0]) for pattern in KV_CACHE_TARGETS]
-        )
-    else:
-        # match on the exact KV_CACHE_TARGETS
-        # if there are multiple targets
-        is_match_targets = set(KV_CACHE_TARGETS) == set(scheme.targets)
+    for target in scheme.targets:
+        if target in KV_CACHE_TARGETS:
+            return True
-    is_match_output_activations = scheme.output_activations is not None
-    return is_match_targets and is_match_output_activations
+    return False
 def parse_out_kv_cache_args(

{compressed_tensors_nightly-0.6.0.20240925.dist-info → compressed_tensors_nightly-0.6.0.20240928.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: compressed-tensors-nightly
-Version: 0.6.0.20240925
+Version: 0.6.0.20240928
 Summary: Library for utilization of compressed safetensors of neural network models
 Home-page: https://github.com/neuralmagic/compressed-tensors
 Author: Neuralmagic, Inc.

{compressed_tensors_nightly-0.6.0.20240925.dist-info → compressed_tensors_nightly-0.6.0.20240928.dist-info}/RECORD RENAMED Viewed

@@ -1,12 +1,12 @@
 compressed_tensors/__init__.py,sha256=UtKmifNeBCSE2TZSAfduVNNzHY-3V7bLjZ7n7RuXLOE,812
-compressed_tensors/base.py,sha256=Mq4mfVQcJhNpha-BXzpOfpmFIdl01o09BJE7D2oQ_00,796
+compressed_tensors/base.py,sha256=7fdFGo8lxjLvrsbBEn0KqceGzcI4RdMSTh8mR6J1Hws,833
 compressed_tensors/version.py,sha256=83tBdwNu2sUhiLPvv6tRNh4Y7u70sZ1TFy3ydWctVL8,1586
 compressed_tensors/compressors/__init__.py,sha256=wmX4VnkUTS63xBwK5-6w8FP78bNZpcdcqvf2KOEC5E4,1133
 compressed_tensors/compressors/base.py,sha256=NfVkhq6PRiq2cvAXaUXLoqC_nVYWdSrkE12c9AXYSMo,9956
 compressed_tensors/compressors/dense.py,sha256=xcWECjcRY4INN6jC7vHx5wvUX3NmnKlxA9SVE1A6m2Q,1267
 compressed_tensors/compressors/helpers.py,sha256=k9avlkmeYj6vkOAvl-MgcixtP7ib24SCfhzZ-RusXfw,5403
 compressed_tensors/compressors/marlin_24.py,sha256=e7fGUyZbjUpA5VUMCPxqcYPGNiwoDKupHJaXWCoVKRw,9410
-compressed_tensors/compressors/model_compressor.py,sha256=Wq-NbjtaVOEElDpcjEYun6QFvAIZee8ZAw_wbifuTDA,16793
+compressed_tensors/compressors/model_compressor.py,sha256=3pMfGTTb8bN8PRNCFuH5k0RbP38r8GS_-cPgCkzL9vk,14355
 compressed_tensors/compressors/naive_quantized.py,sha256=z3h3ca5xKCN69mahutxcbzdv-OysiaxaM8P-Qum6zUQ,4823
 compressed_tensors/compressors/pack_quantized.py,sha256=27RVmJ2wg2dvCoawj407HSmKT3VPGJ6ujAMHlT26WlI,7571
 compressed_tensors/compressors/sparse_bitmask.py,sha256=kiDwBlFV0sJGLcIdDYxIiuF64ccgwDfqq1hWRQThYDc,8647
@@ -16,18 +16,19 @@ compressed_tensors/config/dense.py,sha256=NgSxnFCnckU9-iunxEaqiFwqgdO7YYxlWKR74j
 compressed_tensors/config/sparse_bitmask.py,sha256=pZUboRNZTu6NajGOQEFExoPknak5ynVAUeiiYpS1Gt8,1308
 compressed_tensors/linear/__init__.py,sha256=fH6rjBYAxuwrTzBTlTjTgCYNyh6TCvCqajCz4Im4YrA,617
 compressed_tensors/linear/compressed_linear.py,sha256=G0gEFfxLAUsgRcnfSV-PKz1ZBNTVokOauOoup7SE1mw,3210
-compressed_tensors/quantization/__init__.py,sha256=83J5bPB7PavN2TfCoW7_vEDhfYpm4TDrqYO9vdSQ5bk,760
-compressed_tensors/quantization/quant_args.py,sha256=CmyVtjJeHlqCW-7R5Z7tIw6lXUrzCX6Y9bwgmMxEudY,8069
-compressed_tensors/quantization/quant_config.py,sha256=NpVu8YJ4Xw2pIQW_PGaNaml8kx1bUnxkvb0jBYWbKdE,9971
+compressed_tensors/quantization/__init__.py,sha256=nWP_fsl6Nn0ksEgZPzerGiETdvF-ZfNwPnwGlRiR5pY,805
+compressed_tensors/quantization/cache.py,sha256=vnBB5zasO_XpHomZvzUPVVbzyCz2VgebsHePm0kANzY,6831
+compressed_tensors/quantization/quant_args.py,sha256=73KevZXHyrkMCT_3CxbYHz70fI3i-wcF8NvN0wsBPK4,8271
+compressed_tensors/quantization/quant_config.py,sha256=xcCLkPomAOfjB1X8PmQTw1Bmqs8_JF52dSQ9W07VQZc,10119
 compressed_tensors/quantization/quant_scheme.py,sha256=2ITawuNf76E1CDYBWrfpMP8tyZFykzwU99-eD-WggsM,5930
 compressed_tensors/quantization/lifecycle/__init__.py,sha256=MXE2E7GfIfRRfhrdGy2Og3AZOz5N59B0ZGFcsD89y6c,821
-compressed_tensors/quantization/lifecycle/apply.py,sha256=uftWFunr_CpCZM_qWfo2O1USXKB2qSYD1pBJsO8BuCU,15285
+compressed_tensors/quantization/lifecycle/apply.py,sha256=_rd56GZZkhbu0HWiq6iYzgcnkMsX3GCs-e8DvtmWmbQ,15668
 compressed_tensors/quantization/lifecycle/calibration.py,sha256=PlS_EqCOPqJD3QKuLPXO9AOtDzXtQWvEBTynFv-FFVw,2698
 compressed_tensors/quantization/lifecycle/compressed.py,sha256=Fj9n66IN0EWsOAkBHg3O0GlOQpxstqjCcs0ttzMXrJ0,2296
-compressed_tensors/quantization/lifecycle/forward.py,sha256=PljD9pzATILEOiC3ZdHUTsfSbZdAa6iSIxWmvAHLG9I,13688
-compressed_tensors/quantization/lifecycle/frozen.py,sha256=h1XYt89MouBTf3jTYLG_6OdFxIu5q2N8tPjsy6J4E6Y,1726
+compressed_tensors/quantization/lifecycle/forward.py,sha256=eLup6QDRUUp_Ozcas7RDRLIXBWjFbxn5gWbcAIJEGlw,15715
+compressed_tensors/quantization/lifecycle/frozen.py,sha256=NiJw7NP7pcT6idWFa8vksgiLoT8oQ975e57S4QfD2QQ,1874
 compressed_tensors/quantization/lifecycle/helpers.py,sha256=TmLY_G5VP_Fg2Ywio_dxoHRTxOKZdT7_aG5S9WtD4zI,2424
-compressed_tensors/quantization/lifecycle/initialize.py,sha256=S5Kwy16Da8WUIIpa1xVKc72MijJ5C_rqM6JjanZ7MGk,7133
+compressed_tensors/quantization/lifecycle/initialize.py,sha256=HAtSm7vKOZ3kGZuWe2B8LsmfC5B5vIKlc0V8C4rAF4Y,8819
 compressed_tensors/quantization/observers/__init__.py,sha256=4Sa7rqi5RB_S5bPO8KmncETiqDsoMBhwP37arlQym8s,764
 compressed_tensors/quantization/observers/base.py,sha256=5ovQicWPYHjIxr6-EkQ4lgOX0PpI9g23iSzKpxjM1Zg,8420
 compressed_tensors/quantization/observers/helpers.py,sha256=s_A23Qa_BLfOdHJCN5bm-qPWkhjjj_RIVrhSp1Y9Dtk,4211
@@ -35,7 +36,7 @@ compressed_tensors/quantization/observers/memoryless.py,sha256=jH_c6K3gxf4W3VNXQ
 compressed_tensors/quantization/observers/min_max.py,sha256=sQXqU3z-voxIDfR_9mQzwQUflZj2sASm_G8CYaXntFw,3865
 compressed_tensors/quantization/observers/mse.py,sha256=Aeh-253Vbab1F8cYuBiGNn4OXWJ67wXQ_JVfl3mu2a8,6034
 compressed_tensors/quantization/utils/__init__.py,sha256=VdtEmP0bvuND_IGQnyqUPc5lnFp-1_yD7StKSX4x80w,656
-compressed_tensors/quantization/utils/helpers.py,sha256=pwvU613XRvMDtI5b39II5jukBl5OUCqoX0ofVRpOFRY,8633
+compressed_tensors/quantization/utils/helpers.py,sha256=y4LEyC2oUd876ZMdALWKGH3Ct5EgBJZV4id_NUjTGH8,9531
 compressed_tensors/registry/__init__.py,sha256=FwLSNYqfIrb5JD_6OK_MT4_svvKTN_nEhpgQlQvGbjI,658
 compressed_tensors/registry/registry.py,sha256=fxjOjh2wklCvJhQxwofdy-zV8q7MkQ85SLG77nml2iA,11890
 compressed_tensors/utils/__init__.py,sha256=gS4gSU2pwcAbsKj-6YMaqhm25udFy6ISYaWBf-myRSM,808
@@ -45,8 +46,8 @@ compressed_tensors/utils/permutations_24.py,sha256=kx6fsfDHebx94zsSzhXGyCyuC9sVy
 compressed_tensors/utils/permute.py,sha256=V6tJLKo3Syccj-viv4F7ZKZgJeCB-hl-dK8RKI_kBwI,2355
 compressed_tensors/utils/safetensors_load.py,sha256=m08ANVuTBxQdoa6LufDgcNJ7wCLDJolyZljB8VEybAU,8578
 compressed_tensors/utils/semi_structured_conversions.py,sha256=XKNffPum54kPASgqKzgKvyeqWPAkair2XEQXjkp7ho8,13489
-compressed_tensors_nightly-0.6.0.20240925.dist-info/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
-compressed_tensors_nightly-0.6.0.20240925.dist-info/METADATA,sha256=AHeC-ko08CtK8_xQUnuNlWNQIhmDcKzDpihAiMBHjR8,6799
-compressed_tensors_nightly-0.6.0.20240925.dist-info/WHEEL,sha256=eOLhNAGa2EW3wWl_TU484h7q1UNgy0JXjjoqKoxAAQc,92
-compressed_tensors_nightly-0.6.0.20240925.dist-info/top_level.txt,sha256=w2i-GyPs2s1UwVxvutSvN_lM22SXC2hQFBmoMcPnV7Y,19
-compressed_tensors_nightly-0.6.0.20240925.dist-info/RECORD,,
+compressed_tensors_nightly-0.6.0.20240928.dist-info/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
+compressed_tensors_nightly-0.6.0.20240928.dist-info/METADATA,sha256=vndAZXPsHUGFnoR1oLqalmP1tnMaAUx7QgXHPVrwarE,6799
+compressed_tensors_nightly-0.6.0.20240928.dist-info/WHEEL,sha256=eOLhNAGa2EW3wWl_TU484h7q1UNgy0JXjjoqKoxAAQc,92
+compressed_tensors_nightly-0.6.0.20240928.dist-info/top_level.txt,sha256=w2i-GyPs2s1UwVxvutSvN_lM22SXC2hQFBmoMcPnV7Y,19
+compressed_tensors_nightly-0.6.0.20240928.dist-info/RECORD,,

{compressed_tensors_nightly-0.6.0.20240925.dist-info → compressed_tensors_nightly-0.6.0.20240928.dist-info}/LICENSE RENAMED Viewed

File without changes

{compressed_tensors_nightly-0.6.0.20240925.dist-info → compressed_tensors_nightly-0.6.0.20240928.dist-info}/WHEEL RENAMED Viewed

File without changes

{compressed_tensors_nightly-0.6.0.20240925.dist-info → compressed_tensors_nightly-0.6.0.20240928.dist-info}/top_level.txt RENAMED Viewed

File without changes

compressed-tensors-nightly 0.6.0.20240925__py3-none-any.whl → 0.6.0.20240928__py3-none-any.whl

compressed-tensors-nightly 0.6.0.20240925py3-none-any.whl → 0.6.0.20240928py3-none-any.whl