PyPI - nexaai - Versions diffs - 1.0.19rc7__cp310-cp310-macosx_14_0_universal2.whl → 1.0.19rc8__cp310-cp310-macosx_14_0_universal2.whl - Mend

nexaai 1.0.19rc7__cp310-cp310-macosx_14_0_universal2.whl → 1.0.19rc8__cp310-cp310-macosx_14_0_universal2.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.

This version of nexaai might be problematic. Click here for more details.

Files changed (196) hide show

nexaai/binds/nexa_mlx/py-lib/vlm/modeling/models/idefics3/vision.py DELETED Viewed

@@ -1,233 +0,0 @@
-import inspect
-from dataclasses import dataclass
-from typing import Optional
-import mlx.core as mx
-import mlx.nn as nn
-import numpy as np
-@dataclass
-class VisionConfig:
-    model_type: str
-    hidden_size: int
-    num_attention_heads: int
-    patch_size: int
-    num_hidden_layers: int = 12
-    intermediate_size: int = 3072
-    image_size: int = 224
-    num_channels: int = 3
-    layer_norm_eps: float = 1e-6
-    @classmethod
-    def from_dict(cls, params):
-        return cls(
-            **{
-                k: v
-                for k, v in params.items()
-                if k in inspect.signature(cls).parameters
-            }
-        )
-def check_array_shape(arr):
-    shape = arr.shape
-    # Check if the shape has 4 dimensions
-    if len(shape) != 4:
-        return False
-    out_channels, kH, KW, _ = shape
-    # Check if out_channels is the largest, and kH and KW are the same
-    if (out_channels >= kH) and (out_channels >= KW) and (kH == KW):
-        return True
-    else:
-        return False
-class Attention(nn.Module):
-    def __init__(
-        self,
-        dims: int,
-        num_heads: int,
-        query_input_dims: Optional[int] = None,
-        key_input_dims: Optional[int] = None,
-        value_input_dims: Optional[int] = None,
-        value_dims: Optional[int] = None,
-        value_output_dims: Optional[int] = None,
-        bias: bool = True,
-    ):
-        super().__init__()
-        if (dims % num_heads) != 0:
-            raise ValueError(
-                "The input feature dimensions should be divisible by the "
-                f"number of heads ({dims} % {num_heads}) != 0"
-            )
-        query_input_dims = query_input_dims or dims
-        key_input_dims = key_input_dims or dims
-        value_input_dims = value_input_dims or key_input_dims
-        value_dims = value_dims or dims
-        value_output_dims = value_output_dims or dims
-        self.num_heads = num_heads
-        head_dim = dims // num_heads
-        self.scale = head_dim**-0.5
-        self.q_proj = nn.Linear(query_input_dims, dims, bias=bias)
-        self.k_proj = nn.Linear(key_input_dims, dims, bias=bias)
-        self.v_proj = nn.Linear(value_input_dims, value_dims, bias=bias)
-        self.out_proj = nn.Linear(value_dims, value_output_dims, bias=bias)
-    def __call__(self, x, mask=None):
-        queries = self.q_proj(x)
-        keys = self.k_proj(x)
-        values = self.v_proj(x)
-        num_heads = self.num_heads
-        B, L, D = queries.shape
-        _, S, _ = keys.shape
-        queries = queries.reshape(B, L, num_heads, -1).transpose(0, 2, 1, 3)
-        keys = keys.reshape(B, S, num_heads, -1).transpose(0, 2, 1, 3)
-        values = values.reshape(B, S, num_heads, -1).transpose(0, 2, 1, 3)
-        output = mx.fast.scaled_dot_product_attention(
-            queries, keys, values, scale=self.scale, mask=mask
-        )
-        output = output.transpose(0, 2, 1, 3).reshape(B, L, -1)
-        return self.out_proj(output)
-class MLP(nn.Module):
-    def __init__(self, config: VisionConfig):
-        super().__init__()
-        self.activation_fn = nn.GELU(approx="precise")
-        self.fc1 = nn.Linear(config.hidden_size, config.intermediate_size, bias=True)
-        self.fc2 = nn.Linear(config.intermediate_size, config.hidden_size, bias=True)
-    def __call__(self, x: mx.array) -> mx.array:
-        x = self.fc1(x)
-        x = self.activation_fn(x)
-        x = self.fc2(x)
-        return x
-class EncoderLayer(nn.Module):
-    def __init__(self, config: VisionConfig):
-        super().__init__()
-        self.embed_dim = config.hidden_size
-        self.self_attn = Attention(
-            config.hidden_size, config.num_attention_heads, bias=True
-        )
-        self.layer_norm1 = nn.LayerNorm(self.embed_dim, eps=config.layer_norm_eps)
-        self.mlp = MLP(config)
-        self.layer_norm2 = nn.LayerNorm(self.embed_dim, eps=config.layer_norm_eps)
-    def __call__(self, x: mx.array, mask: Optional[mx.array] = None) -> mx.array:
-        r = self.self_attn(self.layer_norm1(x), mask)
-        h = x + r
-        r = self.mlp(self.layer_norm2(h))
-        return h + r
-class Encoder(nn.Module):
-    def __init__(self, config: VisionConfig):
-        super().__init__()
-        self.layers = [EncoderLayer(config) for _ in range(config.num_hidden_layers)]
-    def __call__(
-        self,
-        x: mx.array,
-        output_hidden_states: Optional[bool] = None,
-        mask: Optional[mx.array] = None,
-    ) -> mx.array:
-        encoder_states = (x,) if output_hidden_states else None
-        h = x
-        for l in self.layers:
-            x = l(x, mask=mask)
-            if output_hidden_states:
-                encoder_states = encoder_states + (x,)
-            h = x
-        return (h, encoder_states)
-class VisionEmbeddings(nn.Module):
-    def __init__(self, config: VisionConfig):
-        super().__init__()
-        self.config = config
-        self.embed_dim = config.hidden_size
-        self.image_size = config.image_size
-        self.patch_size = config.patch_size
-        self.patch_embedding = nn.Conv2d(
-            in_channels=config.num_channels,
-            out_channels=self.embed_dim,
-            kernel_size=self.patch_size,
-            stride=self.patch_size,
-        )
-        self.num_patches = (self.image_size // self.patch_size) ** 2
-        self.num_positions = self.num_patches
-        self.position_embedding = nn.Embedding(self.num_positions, self.embed_dim)
-    def __call__(self, x: mx.array) -> mx.array:
-        patch_embeddings = self.patch_embedding(x)
-        patch_embeddings = mx.flatten(patch_embeddings, start_axis=1, end_axis=2)
-        position_ids = mx.array(mx.arange(self.num_positions)[None, :])
-        embeddings = patch_embeddings
-        embeddings += self.position_embedding(position_ids)
-        return embeddings
-class VisionModel(nn.Module):
-    def __init__(self, config: VisionConfig):
-        super().__init__()
-        self.model_type = config.model_type
-        if self.model_type not in [
-            "siglip_vision_model",
-            "idefics3",
-            "idefics3_vision",
-            "smolvlm_vision",
-        ]:
-            raise ValueError(f"Unsupported model type: {self.model_type}")
-        self.embeddings = VisionEmbeddings(config)
-        self.encoder = Encoder(config)
-        self.post_layernorm = nn.LayerNorm(config.hidden_size)
-    def __call__(
-        self,
-        x: mx.array,
-        output_hidden_states: Optional[bool] = None,
-    ) -> mx.array:
-        x = self.embeddings(x)
-        x = x.astype(self.embeddings.patch_embedding.weight.dtype)
-        encoder_outputs = self.encoder(
-            x=x, output_hidden_states=output_hidden_states, mask=None
-        )
-        pooler_output = self.post_layernorm(encoder_outputs[0])
-        return pooler_output, x, encoder_outputs[-1]
-    def sanitize(self, weights):
-        sanitized_weights = {}
-        for k, v in weights.items():
-            if "position_ids" in k:
-                # Remove unused position_ids
-                continue
-            elif "patch_embedding.weight" in k:
-                # PyTorch conv2d weight tensors have shape:
-                #   [out_channels, in_channels, kH, KW]
-                # MLX conv2d expects the weight be of shape:
-                #   [out_channels, kH, KW, in_channels]
-                if check_array_shape(v):
-                    sanitized_weights[k] = v
-                else:
-                    sanitized_weights[k] = v.transpose(0, 2, 3, 1)
-            else:
-                sanitized_weights[k] = v
-        return sanitized_weights

nexaai/binds/nexa_mlx/py-lib/vlm/modeling/models/internvl_chat/__init__.py DELETED Viewed

@@ -1,9 +0,0 @@
-from .internvl_chat import (
-    LanguageModel,
-    Model,
-    ModelConfig,
-    TextConfig,
-    VisionConfig,
-    VisionModel,
-)
-from .processor import InternVLChatProcessor, InternVLImageProcessor

nexaai/binds/nexa_mlx/py-lib/vlm/modeling/models/internvl_chat/internvl_chat.py DELETED Viewed

@@ -1,140 +0,0 @@
-import glob
-import inspect
-import json
-from dataclasses import dataclass
-from pathlib import Path
-from typing import List, Optional
-import mlx.core as mx
-import mlx.nn as nn
-import numpy as np
-from huggingface_hub import snapshot_download
-from ..base import pixel_shuffle
-from .language import LanguageModel, TextConfig
-from .vision import VisionConfig, VisionModel
-@dataclass
-class ModelConfig:
-    text_config: TextConfig
-    vision_config: VisionConfig
-    model_type: str
-    ignore_index: int = -100
-    image_token_index: int = 151667
-    video_token_index: int = 151656
-    vision_feature_select_strategy: str = "default"
-    vision_feature_layer: int = -1
-    vocab_size: int = 32000
-    downsample_ratio: float = 0.5
-    eos_token_id: Optional[List[int]] = None
-    @classmethod
-    def from_dict(cls, params):
-        return cls(
-            **{
-                k: v
-                for k, v in params.items()
-                if k in inspect.signature(cls).parameters
-            }
-        )
-class Model(nn.Module):
-    def __init__(self, config: ModelConfig):
-        super().__init__()
-        self.config = config
-        self.vision_model = VisionModel(config.vision_config)
-        self.language_model = LanguageModel(config.text_config)
-        self.downsample_ratio = config.downsample_ratio
-        vit_hidden_size = self.config.vision_config.hidden_size
-        llm_hidden_size = self.config.text_config.hidden_size
-        self.mlp1 = [
-            nn.LayerNorm(vit_hidden_size * int(1 / self.downsample_ratio) ** 2),
-            nn.Linear(
-                vit_hidden_size * int(1 / self.downsample_ratio) ** 2, llm_hidden_size
-            ),
-            nn.GELU(),
-            nn.Linear(llm_hidden_size, llm_hidden_size),
-        ]
-    def get_input_embeddings(
-        self,
-        input_ids: Optional[mx.array] = None,
-        pixel_values: Optional[mx.array] = None,
-    ):
-        if pixel_values is None:
-            return self.language_model.model.embed_tokens(input_ids)
-        dtype = self.vision_model.embeddings.patch_embedding.weight.dtype
-        pixel_values = pixel_values.astype(dtype)
-        # TODO: Remove this after transformers implementation is merged
-        if pixel_values.ndim == 5:
-            pixel_values = pixel_values[0]
-        # Get the input embeddings from the language model
-        inputs_embeds = self.language_model.model.embed_tokens(input_ids)
-        # Get the ouptut hidden states from the vision model
-        hidden_states, _, _ = self.vision_model(
-            pixel_values.transpose(0, 2, 3, 1), output_hidden_states=True
-        )
-        # Extract vision embeddings, removing the class token (first token)
-        hidden_states = hidden_states[:, 1:, :]
-        # Apply pixel shuffle with downsampling
-        hidden_states = pixel_shuffle(
-            hidden_states, shuffle_ratio=self.downsample_ratio
-        )
-        # Apply MLP transformation
-        for layer in self.mlp1:
-            hidden_states = layer(hidden_states)
-        # Insert special image tokens in the input_ids
-        final_inputs_embeds = self._merge_input_ids_with_image_features(
-            hidden_states, inputs_embeds, input_ids
-        )
-        return final_inputs_embeds
-    def _merge_input_ids_with_image_features(
-        self, image_features, inputs_embeds, input_ids
-    ):
-        B, N, C = inputs_embeds.shape
-        image_token_index = self.config.image_token_index
-        video_token_index = self.config.video_token_index
-        # Positions of <image> tokens in input_ids, assuming batch size is 1
-        image_positions = input_ids == image_token_index
-        if mx.sum(image_positions) == 0:
-            image_positions = input_ids == video_token_index
-        image_indices = np.where(image_positions)[1].tolist()
-        image_features = image_features.reshape(-1, image_features.shape[-1])
-        inputs_embeds[:, image_indices, :] = image_features
-        return inputs_embeds.reshape(B, N, C)
-    @property
-    def layers(self):
-        return self.language_model.model.layers
-    def __call__(
-        self,
-        input_ids: mx.array,
-        pixel_values: mx.array,
-        mask: mx.array,
-        cache=None,
-        **kwargs,
-    ):
-        input_embddings = self.get_input_embeddings(input_ids, pixel_values)
-        logits = self.language_model(None, cache=cache, inputs_embeds=input_embddings)
-        return logits

nexaai/binds/nexa_mlx/py-lib/vlm/modeling/models/internvl_chat/language.py DELETED Viewed

@@ -1,220 +0,0 @@
-import inspect
-from dataclasses import dataclass
-from typing import Dict, Optional, Union
-import mlx.core as mx
-import mlx.nn as nn
-from ..base import (
-    LanguageModelOutput,
-    create_attention_mask,
-    scaled_dot_product_attention,
-)
-from ..cache import KVCache
-@dataclass
-class TextConfig:
-    model_type: str
-    hidden_size: int
-    num_hidden_layers: int
-    intermediate_size: int
-    num_attention_heads: int
-    rms_norm_eps: float
-    vocab_size: int
-    max_window_layers: int
-    hidden_act: str
-    num_key_value_heads: Optional[int] = 8
-    max_position_embeddings: Optional[int] = 40960
-    rope_theta: float = 1000000.0
-    rope_traditional: bool = False
-    rope_scaling: Optional[Dict[str, Union[float, str]]] = None
-    tie_word_embeddings: bool = False
-    sliding_window: int = 32768
-    use_sliding_window: bool = False
-    use_cache: bool = True
-    def __post_init__(self):
-        if self.num_key_value_heads is None:
-            self.num_key_value_heads = self.num_attention_heads
-    @classmethod
-    def from_dict(cls, params):
-        return cls(
-            **{
-                k: v
-                for k, v in params.items()
-                if k in inspect.signature(cls).parameters
-            }
-        )
-class Attention(nn.Module):
-    def __init__(self, args: TextConfig):
-        super().__init__()
-        dim = args.hidden_size
-        self.n_heads = n_heads = args.num_attention_heads
-        assert args.num_key_value_heads is not None
-        self.n_kv_heads = n_kv_heads = args.num_key_value_heads
-        self.head_dim = head_dim = args.hidden_size // n_heads
-        self.scale = head_dim**-0.5
-        self.q_proj = nn.Linear(dim, n_heads * head_dim, bias=True)
-        self.k_proj = nn.Linear(dim, n_kv_heads * head_dim, bias=True)
-        self.v_proj = nn.Linear(dim, n_kv_heads * head_dim, bias=True)
-        self.o_proj = nn.Linear(n_heads * head_dim, dim, bias=False)
-        self.rotary_emb = nn.RoPE(
-            head_dim,
-            base=args.rope_theta,
-            traditional=args.rope_traditional,
-        )
-    def __call__(
-        self,
-        x: mx.array,
-        mask: Optional[mx.array] = None,
-        cache: Optional[KVCache] = None,
-    ) -> mx.array:
-        B, L, D = x.shape
-        queries, keys, values = self.q_proj(x), self.k_proj(x), self.v_proj(x)
-        # Prepare the queries, keys and values for the attention computation
-        queries = queries.reshape(B, L, self.n_heads, self.head_dim).transpose(
-            0, 2, 1, 3
-        )
-        keys = keys.reshape(B, L, self.n_kv_heads, self.head_dim).transpose(0, 2, 1, 3)
-        values = values.reshape(B, L, self.n_kv_heads, self.head_dim).transpose(
-            0, 2, 1, 3
-        )
-        offset = cache.offset if cache else 0
-        if mask is not None and isinstance(mask, mx.array):
-            mask = mask[..., : keys.shape[-2]]
-        queries = self.rotary_emb(queries, offset=offset)
-        keys = self.rotary_emb(keys, offset=offset)
-        if cache is not None:
-            keys, values = cache.update_and_fetch(keys, values)
-        output = scaled_dot_product_attention(
-            queries, keys, values, cache, scale=self.scale, mask=mask
-        )
-        output = output.transpose(0, 2, 1, 3).reshape(B, L, -1)
-        return self.o_proj(output)
-class MLP(nn.Module):
-    def __init__(self, dim, hidden_dim):
-        super().__init__()
-        self.gate_proj = nn.Linear(dim, hidden_dim, bias=False)
-        self.down_proj = nn.Linear(hidden_dim, dim, bias=False)
-        self.up_proj = nn.Linear(dim, hidden_dim, bias=False)
-    def __call__(self, x) -> mx.array:
-        return self.down_proj(nn.silu(self.gate_proj(x)) * self.up_proj(x))
-class Qwen2VLDecoderLayer(nn.Module):
-    def __init__(self, args: TextConfig):
-        super().__init__()
-        self.num_attention_heads = args.num_attention_heads
-        self.hidden_size = args.hidden_size
-        self.self_attn = Attention(args)
-        self.mlp = MLP(args.hidden_size, args.intermediate_size)
-        self.input_layernorm = nn.RMSNorm(args.hidden_size, eps=args.rms_norm_eps)
-        self.post_attention_layernorm = nn.RMSNorm(
-            args.hidden_size, eps=args.rms_norm_eps
-        )
-        self.args = args
-    def __call__(
-        self,
-        x: mx.array,
-        mask: Optional[mx.array] = None,
-        cache: Optional[KVCache] = None,
-    ) -> mx.array:
-        r = self.self_attn(self.input_layernorm(x), mask, cache)
-        h = x + r
-        r = self.mlp(self.post_attention_layernorm(h))
-        out = h + r
-        return out
-class Qwen2Model(nn.Module):
-    def __init__(self, args: TextConfig):
-        super().__init__()
-        self.args = args
-        self.vocab_size = args.vocab_size
-        self.num_hidden_layers = args.num_hidden_layers
-        assert self.vocab_size > 0
-        self.embed_tokens = nn.Embedding(args.vocab_size, args.hidden_size)
-        self.layers = [
-            Qwen2VLDecoderLayer(args=args) for _ in range(args.num_hidden_layers)
-        ]
-        self.norm = nn.RMSNorm(args.hidden_size, eps=args.rms_norm_eps)
-    def __call__(
-        self,
-        inputs: mx.array,
-        inputs_embeds: Optional[mx.array] = None,
-        mask: Optional[mx.array] = None,
-        cache=None,
-    ):
-        if inputs_embeds is None:
-            h = self.embed_tokens(inputs)
-        else:
-            h = inputs_embeds
-        if cache is None:
-            cache = [None] * len(self.layers)
-        if mask is None:
-            mask = create_attention_mask(h, cache)
-        for layer, c in zip(self.layers, cache):
-            h = layer(h, mask, c)
-        return self.norm(h)
-class LanguageModel(nn.Module):
-    def __init__(self, args: TextConfig):
-        super().__init__()
-        self.args = args
-        self.model_type = args.model_type
-        self.model = Qwen2Model(args)
-        if not args.tie_word_embeddings:
-            self.lm_head = nn.Linear(args.hidden_size, args.vocab_size, bias=False)
-    def __call__(
-        self,
-        inputs: mx.array,
-        inputs_embeds: Optional[mx.array] = None,
-        mask: Optional[mx.array] = None,
-        cache=None,
-    ):
-        out = self.model(inputs, cache=cache, inputs_embeds=inputs_embeds)
-        if self.args.tie_word_embeddings:
-            out = self.model.embed_tokens.as_linear(out)
-        else:
-            out = self.lm_head(out)
-        return LanguageModelOutput(logits=out)
-    @property
-    def layers(self):
-        return self.model.layers
-    @property
-    def head_dim(self):
-        return self.args.hidden_size // self.args.num_attention_heads
-    @property
-    def n_kv_heads(self):
-        return self.args.num_key_value_heads