PyPI - opentau - Versions diffs - 0.1.0__tar.gz → 0.1.2__tar.gz - Mend

opentau 0.1.0tar.gz → 0.1.2tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (117) hide show

{opentau-0.1.0/src/opentau.egg-info → opentau-0.1.2}/PKG-INFO RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.4
 Name: opentau
-Version: 0.1.0
+Version: 0.1.2
 Summary: OpenTau: Tensor's VLA Training Infrastructure for Real-World Robotics in Pytorch
 Author-email: Shuheng Liu <wish1104@icloud.com>, William Yue <williamyue37@gmail.com>, Akshay Shah <akshayhitendrashah@gmail.com>, Xingrui Gu <xingrui_gu@berkeley.edu>
 License: Apache-2.0
@@ -52,7 +52,7 @@ Requires-Dist: onnxruntime>=1.22.1; sys_platform == "darwin" or platform_machine
 Requires-Dist: onnxruntime-gpu>=1.22.0; (sys_platform == "linux" and platform_machine == "x86_64") or (sys_platform == "win32" and (platform_machine == "AMD64" or platform_machine == "x86_64"))
 Requires-Dist: onnxscript>=0.3.1
 Requires-Dist: onnx-ir>=0.1.4
-Requires-Dist: opentau-transformers==4.53.3
+Requires-Dist: transformers==4.53.3
 Requires-Dist: scipy>=1.15.2
 Requires-Dist: pytest>=8.1.0
 Requires-Dist: pytest-cov>=5.0.0
@@ -94,6 +94,9 @@ Requires-Dist: numpy<2; extra == "libero"
 Requires-Dist: gym<0.27,>=0.25; extra == "libero"
 Requires-Dist: pyopengl-accelerate==3.1.7; sys_platform == "linux" and extra == "libero"
 Requires-Dist: gymnasium[other]>=0.29; extra == "libero"
+Requires-Dist: mujoco>=3.1.6; sys_platform == "linux" and extra == "libero"
+Requires-Dist: pyopengl==3.1.7; sys_platform == "linux" and extra == "libero"
+Requires-Dist: numpy==1.26.4; sys_platform == "linux" and extra == "libero"
 Dynamic: license-file
 <p align="center">
@@ -134,10 +137,10 @@ OpenTau ($\tau$) is a tool developed by *[Tensor][1]* to bridge this gap, and we
 ## Quick Start
 If you are familiar with LeRobot, getting started with OpenTau is very easy.
 Because OpenTau is a fork of the popular LeRobot repository, any LeRobot-compliant policy and dataset can be used directly with OpenTau.
-Check out our documentation to get started quickly.
-We provide a quick start guide to help you get started with OpenTau.
+Check out our [documentation](https://opentau.readthedocs.io/) to get started quickly.
+We provide a [quick start guide](https://opentau.readthedocs.io/en/latest/getting_started.html) to help you get started with OpenTau.
-For using local notebooks to train and evaluate models, find the notebooks at `notebooks/pi05_training.ipynb` and `notebooks/pi05_evaluation_only.ipynb`.
+For using local notebooks to train and evaluate models, find the notebooks at [notebooks/pi05_training.ipynb](https://github.com/TensorAuto/OpenTau/blob/main/notebooks/pi05_training.ipynb) and [notebooks/pi05_evaluation_only.ipynb](https://github.com/TensorAuto/OpenTau/blob/main/notebooks/pi05_evaluation_only.ipynb).
 For using the Google Colab notebooks to train and evaluate models, find the colab notebooks here: [pi05_training](https://colab.research.google.com/drive/1DeU0lNnEzs1KHo0Nkgh4YKBr-xu9moBM?usp=sharing) and [pi05_evaluation_only](https://colab.research.google.com/drive/1U_AyuH9WYMT4anEWvsOtIT7g01jA0WGm?usp=sharing) respectively.

{opentau-0.1.0 → opentau-0.1.2}/README.md RENAMED Viewed

@@ -36,10 +36,10 @@ OpenTau ($\tau$) is a tool developed by *[Tensor][1]* to bridge this gap, and we
 ## Quick Start
 If you are familiar with LeRobot, getting started with OpenTau is very easy.
 Because OpenTau is a fork of the popular LeRobot repository, any LeRobot-compliant policy and dataset can be used directly with OpenTau.
-Check out our documentation to get started quickly.
-We provide a quick start guide to help you get started with OpenTau.
+Check out our [documentation](https://opentau.readthedocs.io/) to get started quickly.
+We provide a [quick start guide](https://opentau.readthedocs.io/en/latest/getting_started.html) to help you get started with OpenTau.
-For using local notebooks to train and evaluate models, find the notebooks at `notebooks/pi05_training.ipynb` and `notebooks/pi05_evaluation_only.ipynb`.
+For using local notebooks to train and evaluate models, find the notebooks at [notebooks/pi05_training.ipynb](https://github.com/TensorAuto/OpenTau/blob/main/notebooks/pi05_training.ipynb) and [notebooks/pi05_evaluation_only.ipynb](https://github.com/TensorAuto/OpenTau/blob/main/notebooks/pi05_evaluation_only.ipynb).
 For using the Google Colab notebooks to train and evaluate models, find the colab notebooks here: [pi05_training](https://colab.research.google.com/drive/1DeU0lNnEzs1KHo0Nkgh4YKBr-xu9moBM?usp=sharing) and [pi05_evaluation_only](https://colab.research.google.com/drive/1U_AyuH9WYMT4anEWvsOtIT7g01jA0WGm?usp=sharing) respectively.

{opentau-0.1.0 → opentau-0.1.2}/pyproject.toml RENAMED Viewed

@@ -20,7 +20,7 @@ huggingface = "https://huggingface.co/TensorAuto"
 [project]
 name = "opentau"
-version = "0.1.0"
+version = "0.1.2"
 description = "OpenTau: Tensor's VLA Training Infrastructure for Real-World Robotics in Pytorch"
 authors = [
     { name = "Shuheng Liu", email = "wish1104@icloud.com" },
@@ -76,7 +76,7 @@ dependencies = [
     "onnxruntime-gpu>=1.22.0 ; ((sys_platform == 'linux' and platform_machine == 'x86_64') or (sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64'))) ",
     "onnxscript>=0.3.1",
     "onnx-ir>=0.1.4",
-    "opentau-transformers==4.53.3",
+    "transformers==4.53.3",
     "scipy>=1.15.2",
     "pytest>=8.1.0",
     "pytest-cov>=5.0.0",
@@ -89,7 +89,10 @@ dependencies = [
 ]
 [project.scripts]
-opentau-train = "opentau.scripts.launch_train:main"
+opentau-train = "opentau.scripts.launch:train"
+opentau-eval = "opentau.scripts.launch:eval"
+opentau-export = "opentau.scripts.launch:export"
+opentau-dataset-viz = "opentau.scripts.launch:visualize"
 [project.optional-dependencies]
 dev = ["pre-commit>=3.7.0",
@@ -122,6 +125,9 @@ libero = [
     "gym>=0.25,<0.27",
     "pyopengl-accelerate==3.1.7 ; sys_platform == 'linux'",
     "gymnasium[other]>=0.29",
+    "mujoco>=3.1.6 ; sys_platform == 'linux'",
+    "pyopengl==3.1.7 ; sys_platform == 'linux'",
+    "numpy==1.26.4 ; sys_platform == 'linux'",
 ]
 [tool.uv.sources]

{opentau-0.1.0 → opentau-0.1.2}/src/opentau/__init__.py RENAMED Viewed

@@ -56,6 +56,7 @@ When implementing a new policy class (e.g., `DiffusionPolicy`), follow these ste
 import itertools
 from opentau.__version__ import __version__  # noqa: F401
+from opentau.utils import transformers_patch  # noqa: F401
 # TODO(rcadene): Improve policies and envs. As of now, an item in `available_policies`
 # refers to a yaml file AND a modeling name. Same for `available_envs` which refers to

{opentau-0.1.0 → opentau-0.1.2}/src/opentau/datasets/lerobot_dataset.py RENAMED Viewed

@@ -633,7 +633,9 @@ class BaseDataset(torch.utils.data.Dataset):
         For example, {"image_key": torch.zeros(2, 3, 224, 224), "image_key_is_pad": [False, True] } will become
         {
             "image_key": torch.zeros(3, 224, 224),
+            "image_key_local": torch.zeros(3, 224, 224),
             "image_key_is_pad: False,
+            "image_key_local_is_pad": True,
         }.
         """
         raise NotImplementedError
@@ -1787,16 +1789,12 @@ class LeRobotDataset(BaseDataset):
         cam_keys = {v for k, v in name_map.items() if k.startswith("camera")}
         for k in cam_keys:
             images = item.pop(k)
-            assert len(images) == 2, (
-                f"{k} in {self.__class__} is expected to have length 2, got shape={images.shape}"
-            )
-            item[k + "_local"], item[k] = images
+            if len(images) == 2:
+                item[k + "_local"], item[k] = images
-            pads = item.pop(k + "_is_pad")
-            assert len(pads) == 2, (
-                f"{k} in {self.__class__} is expected to have length 2, got shape={pads.shape}"
-            )
-            item[k + "_local_is_pad"], item[k + "_is_pad"] = pads
+            pads = item.get(k + "_is_pad")
+            if hasattr(pads, "__len__") and len(pads) == 2:
+                item[k + "_local_is_pad"], item[k + "_is_pad"] = pads
     @staticmethod
     def compute_delta_params(

opentau-0.1.2/src/opentau/scripts/launch.py ADDED Viewed

@@ -0,0 +1,84 @@
+# Copyright 2026 Tensor Auto Inc. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import argparse
+import subprocess
+import sys
+from pathlib import Path
+from types import ModuleType
+import opentau.scripts.eval as eval_script
+import opentau.scripts.export_to_onnx as export_script
+import opentau.scripts.train as train_script
+import opentau.scripts.visualize_dataset as visualize_script
+def launch(script_module: ModuleType, description: str, use_accelerate: bool = True):
+    """Generic launcher for OpenTau scripts using Accelerate or Python."""
+    parser = argparse.ArgumentParser(
+        description=description,
+        usage=f"{Path(sys.argv[0]).name} {'[--accelerate-config CONFIG] ' if use_accelerate else ''}[ARGS]",
+    )
+    if use_accelerate:
+        parser.add_argument(
+            "--accelerate-config", type=str, help="Path to accelerate config file (yaml)", default=None
+        )
+    # We use parse_known_args so that all other arguments are collected
+    # These will be passed to the target script
+    args, unknown_args = parser.parse_known_args()
+    # Base command
+    if use_accelerate:
+        cmd = ["accelerate", "launch"]
+        # Add accelerate config if provided
+        if args.accelerate_config:
+            cmd.extend(["--config_file", args.accelerate_config])
+    else:
+        cmd = [sys.executable]
+    # Add the path to the target script
+    # We resolve the path to ensure it's absolute
+    script_path = Path(script_module.__file__).resolve()
+    cmd.append(str(script_path))
+    # Add all other arguments (passed to the target script)
+    cmd.extend(unknown_args)
+    # Print the command for transparency
+    print(f"Executing: {' '.join(cmd)}")
+    # Replace the current process with the accelerate launch command
+    try:
+        subprocess.run(cmd, check=True)
+    except subprocess.CalledProcessError as e:
+        sys.exit(e.returncode)
+    except KeyboardInterrupt:
+        sys.exit(130)
+def train():
+    launch(train_script, "Launch OpenTau training with Accelerate")
+def eval():
+    launch(eval_script, "Launch OpenTau evaluation with Accelerate")
+def export():
+    launch(export_script, "Launch OpenTau ONNX export", use_accelerate=False)
+def visualize():
+    launch(visualize_script, "Launch OpenTau visualization", use_accelerate=False)

{opentau-0.1.0 → opentau-0.1.2}/src/opentau/scripts/train.py RENAMED Viewed

@@ -73,16 +73,16 @@ def update_policy(
         train_config.loss_weighting["MSE"] * losses["MSE"] + train_config.loss_weighting["CE"] * losses["CE"]
     )
-    # accelerator.backward(loss)
-    # accelerator.unscale_gradients(optimizer=optimizer)
+    accelerator.backward(loss)
+    accelerator.unscale_gradients(optimizer=optimizer)
-    # if accelerator.sync_gradients:
-    #     grad_norm = accelerator.clip_grad_norm_(policy.parameters(), grad_clip_norm)
-    #     if accelerator.is_main_process:
-    #         train_metrics.grad_norm = grad_norm
+    if accelerator.sync_gradients:
+        grad_norm = accelerator.clip_grad_norm_(policy.parameters(), grad_clip_norm)
+        if accelerator.is_main_process:
+            train_metrics.grad_norm = grad_norm
-    # optimizer.step()
-    # optimizer.zero_grad()
+    optimizer.step()
+    optimizer.zero_grad()
     # Step through pytorch scheduler at every batch instead of epoch
     if lr_scheduler is not None:

{opentau-0.1.0 → opentau-0.1.2}/src/opentau/scripts/visualize_dataset.py RENAMED Viewed

@@ -14,7 +14,7 @@
 # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 # See the License for the specific language governing permissions and
 # limitations under the License.
-""" Visualize data of **all** frames of any episode of a dataset of type LeRobotDataset.
+"""Visualize data of **all** frames of any episode of a dataset of type LeRobotDataset.
 Note: The last frame of the episode doesn't always correspond to a final state.
 That's because our datasets are composed of transition from state to state up to
@@ -30,34 +30,21 @@ Examples:
 - Visualize data stored on a local machine:
 ```
-local$ python src/opentau/scripts/visualize_dataset.py \
-    --repo-id lerobot/pusht \
-    --episode-index 0
+local$ opentau-dataset-viz --repo-id lerobot/pusht --episode-index 0
 ```
 - Visualize data stored on a distant machine with a local viewer:
 ```
-distant$ python src/opentau/scripts/visualize_dataset.py \
-    --repo-id lerobot/pusht \
-    --episode-index 0 \
-    --save 1 \
-    --output-dir path/to/directory
+distant$ opentau-dataset-viz --repo-id lerobot/pusht --episode-index 0 --save 1 --output-dir path/to/directory
 local$ scp distant:path/to/directory/lerobot_pusht_episode_0.rrd .
 local$ rerun lerobot_pusht_episode_0.rrd
 ```
 - Visualize data stored on a distant machine through streaming:
-(You need to forward the websocket port to the distant machine, with
-`ssh -L 9087:localhost:9087 username@remote-host`)
 ```
-distant$ python src/opentau/scripts/visualize_dataset.py \
-    --repo-id lerobot/pusht \
-    --episode-index 0 \
-    --mode distant \
-    --ws-port 9087
-local$ rerun ws://localhost:9087
+distant$ opentau-dataset-viz --repo-id lerobot/pusht --episode-index 0 --mode distant --web-port 9090
 ```
 """
@@ -75,8 +62,34 @@ import torch
 import torch.utils.data
 import tqdm
+from opentau.configs.default import DatasetMixtureConfig, WandBConfig
+from opentau.configs.train import TrainPipelineConfig
 from opentau.datasets.lerobot_dataset import LeRobotDataset
-from opentau.scripts.visualize_dataset_html import create_mock_train_config
+def create_mock_train_config() -> TrainPipelineConfig:
+    """Create a mock TrainPipelineConfig for dataset visualization.
+    Returns:
+        TrainPipelineConfig: A mock config with default values.
+    """
+    return TrainPipelineConfig(
+        dataset_mixture=DatasetMixtureConfig(),  # Will be set by the dataset
+        resolution=(224, 224),
+        num_cams=2,
+        max_state_dim=32,
+        max_action_dim=32,
+        action_chunk=50,
+        loss_weighting={"MSE": 1, "CE": 1},
+        num_workers=4,
+        batch_size=8,
+        steps=100_000,
+        log_freq=200,
+        save_checkpoint=True,
+        save_freq=20_000,
+        use_policy_training_preset=True,
+        wandb=WandBConfig(),
+    )
 class EpisodeSampler(torch.utils.data.Sampler):
@@ -108,7 +121,6 @@ def visualize_dataset(
     num_workers: int = 0,
     mode: str = "local",
     web_port: int = 9090,
-    ws_port: int = 9087,
     save: bool = False,
     output_dir: Path | None = None,
 ) -> Path | None:
@@ -142,7 +154,7 @@ def visualize_dataset(
     gc.collect()
     if mode == "distant":
-        rr.serve(open_browser=False, web_port=web_port, ws_port=ws_port)
+        rr.serve_web_viewer(open_browser=False, web_port=web_port)
     logging.info("Logging to Rerun")
@@ -194,7 +206,7 @@ def visualize_dataset(
             print("Ctrl-C received. Exiting.")
-def main():
+def parse_args() -> dict:
     parser = argparse.ArgumentParser()
     parser.add_argument(
@@ -250,12 +262,6 @@ def main():
         default=9090,
         help="Web port for rerun.io when `--mode distant` is set.",
     )
-    parser.add_argument(
-        "--ws-port",
-        type=int,
-        default=9087,
-        help="Web socket port for rerun.io when `--mode distant` is set.",
-    )
     parser.add_argument(
         "--save",
         type=int,
@@ -279,15 +285,25 @@ def main():
     )
     args = parser.parse_args()
-    kwargs = vars(args)
+    return vars(args)
+def main():
+    kwargs = parse_args()
     repo_id = kwargs.pop("repo_id")
     root = kwargs.pop("root")
     tolerance_s = kwargs.pop("tolerance_s")
     logging.info("Loading dataset")
-    dataset = LeRobotDataset(create_mock_train_config(), repo_id, root=root, tolerance_s=tolerance_s)
+    dataset = LeRobotDataset(
+        create_mock_train_config(),
+        repo_id,
+        root=root,
+        tolerance_s=tolerance_s,
+        standardize=False,
+    )
-    visualize_dataset(dataset, **vars(args))
+    visualize_dataset(dataset, **kwargs)
 if __name__ == "__main__":

opentau-0.1.2/src/opentau/utils/transformers_patch.py ADDED Viewed

@@ -0,0 +1,276 @@
+# Copyright 2026 Tensor Auto Inc. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Module for patching transformers
+Most patches come from the branch fix/lerobot-openpi
+"""
+from typing import Optional, Tuple
+import torch
+from torch import nn
+from transformers.models.gemma import modeling_gemma
+from transformers.models.gemma.configuration_gemma import GemmaConfig
+from transformers.models.paligemma.modeling_paligemma import PaliGemmaModel
+# Monkey patch __init__ of GemmaConfig to fix or modify its behavior as needed.
+_original_gemma_config_init = GemmaConfig.__init__
+def patched_gemma_config_init(
+    self, *args, use_adarms: bool = False, adarms_cond_dim: Optional[int] = None, **kwargs
+):
+    """Initializes the GemmaConfig with added ADARMS support.
+    Args:
+        self: The GemmaConfig instance.
+        *args: Variable length argument list.
+        use_adarms: Whether to use Adaptive RMS normalization.
+        adarms_cond_dim: The dimension of the conditioning vector for ADARMS.
+        **kwargs: Arbitrary keyword arguments.
+    """
+    # Call the original init with all other arguments
+    _original_gemma_config_init(self, *args, **kwargs)
+    # Initialize custom attributes
+    self.use_adarms = use_adarms
+    self.adarms_cond_dim = adarms_cond_dim
+    # Set default for adarms_cond_dim if use_adarms is True
+    if self.use_adarms and self.adarms_cond_dim is None:
+        # hidden_size is set by _original_gemma_config_init
+        self.adarms_cond_dim = self.hidden_size
+GemmaConfig.__init__ = patched_gemma_config_init
+# --- Modeling Patches ---
+def _gated_residual(x, y, gate):
+    """
+    Applies gated residual connection with optional gate parameter.
+    Args:
+        x: Input tensor (residual)
+        y: Output tensor to be added
+        gate: Optional gate tensor to modulate the addition
+    Returns:
+        x + y if gate is None, otherwise x + y * gate
+    """
+    if x is None and y is None:
+        return None
+    if x is None or y is None:
+        return x if x is not None else y
+    if gate is None:
+        return x + y
+    return x + y * gate
+modeling_gemma._gated_residual = _gated_residual
+class PatchedGemmaRMSNorm(nn.Module):
+    """RMS normalization with optional adaptive support (ADARMS)."""
+    def __init__(self, dim: int, eps: float = 1e-6, cond_dim: Optional[int] = None):
+        """Initializes the PatchedGemmaRMSNorm.
+        Args:
+            dim: The dimension of the input tensor.
+            eps: The epsilon value for numerical stability.
+            cond_dim: The dimension of the conditioning vector (if using ADARMS).
+        """
+        super().__init__()
+        self.eps = eps
+        self.dim = dim
+        self.cond_dim = cond_dim
+        # Dense layer for adaptive normalization (if cond_dim is provided)
+        if cond_dim is not None:
+            self.dense = nn.Linear(cond_dim, dim * 3, bias=True)
+            # Initialize with zeros (matches source implementation)
+            nn.init.zeros_(self.dense.weight)
+        else:
+            self.weight = nn.Parameter(torch.zeros(dim))
+            self.dense = None
+    def _norm(self, x: torch.Tensor) -> torch.Tensor:
+        """Applies RMS normalization.
+        Args:
+            x: The input tensor.
+        Returns:
+            The normalized tensor.
+        """
+        # Compute variance in float32 (like the source implementation)
+        var = torch.mean(torch.square(x.float()), dim=-1, keepdim=True)
+        # Compute normalization in float32
+        normed_inputs = x * torch.rsqrt(var + self.eps)
+        return normed_inputs
+    def forward(
+        self, x: torch.Tensor, cond: Optional[torch.Tensor] = None
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:
+        """Forward pass of the normalization layer.
+        Args:
+            x: The input tensor.
+            cond: The conditioning tensor for adaptive normalization.
+        Returns:
+            A tuple containing the normalized tensor and the gate tensor (if applicable).
+            If cond is None, the gate tensor will be None.
+        Raises:
+            ValueError: If cond dimension does not match the configured cond_dim.
+        """
+        dtype = x.dtype  # original dtype, could be half-precision
+        normed_inputs = self._norm(x)
+        if cond is None or self.dense is None:
+            # regular RMSNorm
+            # scale by learned parameter in float32 (matches source implementation)
+            normed_inputs = normed_inputs * (1.0 + self.weight.float())
+            return normed_inputs.to(dtype), None  # return in original dtype with None gate
+        # adaptive RMSNorm (if cond is provided and dense layer exists)
+        if cond.shape[-1] != self.cond_dim:
+            raise ValueError(f"Expected cond dimension {self.cond_dim}, got {cond.shape[-1]}")
+        modulation = self.dense(cond)
+        # Reshape modulation to broadcast properly: [batch, 1, features] for [batch, seq, features]
+        if len(x.shape) == 3:  # [batch, seq, features]
+            modulation = modulation.unsqueeze(1)
+        scale, shift, gate = torch.chunk(modulation, 3, dim=-1)
+        normed_inputs = normed_inputs * (1 + scale.to(torch.float32)) + shift.to(torch.float32)
+        return normed_inputs.to(dtype), gate.to(dtype)
+    def extra_repr(self) -> str:
+        """Returns the extra representation of the module."""
+        repr_str = f"{tuple(self.weight.shape)}, eps={self.eps}"
+        if self.dense is not None:
+            repr_str += f", adaptive=True, cond_dim={self.cond_dim}"
+        return repr_str
+# Apply patches
+modeling_gemma.GemmaRMSNorm = PatchedGemmaRMSNorm
+def patched_gemma_decoder_layer_init(self, config: GemmaConfig, layer_idx: int):
+    """Initializes a GemmaDecoderLayer with potential ADARMS support.
+    Args:
+        self: The GemmaDecoderLayer instance.
+        config: The configuration object.
+        layer_idx: The index of the layer.
+    """
+    modeling_gemma.GradientCheckpointingLayer.__init__(self)
+    self.hidden_size = config.hidden_size
+    self.self_attn = modeling_gemma.GemmaAttention(config=config, layer_idx=layer_idx)
+    self.mlp = modeling_gemma.GemmaMLP(config)
+    cond_dim = getattr(config, "adarms_cond_dim", None) if getattr(config, "use_adarms", False) else None
+    self.input_layernorm = modeling_gemma.GemmaRMSNorm(
+        config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim
+    )
+    self.post_attention_layernorm = modeling_gemma.GemmaRMSNorm(
+        config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim
+    )
+modeling_gemma.GemmaDecoderLayer.__init__ = patched_gemma_decoder_layer_init
+def patched_gemma_model_init(self, config: GemmaConfig):
+    """Initializes the GemmaModel with potential ADARMS support.
+    Args:
+        self: The GemmaModel instance.
+        config: The configuration object.
+    """
+    modeling_gemma.GemmaPreTrainedModel.__init__(self, config)
+    self.padding_idx = config.pad_token_id
+    self.vocab_size = config.vocab_size
+    self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size, self.padding_idx)
+    self.layers = nn.ModuleList(
+        [modeling_gemma.GemmaDecoderLayer(config, layer_idx) for layer_idx in range(config.num_hidden_layers)]
+    )
+    cond_dim = getattr(config, "adarms_cond_dim", None) if getattr(config, "use_adarms", False) else None
+    self.norm = modeling_gemma.GemmaRMSNorm(config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim)
+    self.rotary_emb = modeling_gemma.GemmaRotaryEmbedding(config=config)
+    self.gradient_checkpointing = False
+    # Initialize weights and apply final processing
+    self.post_init()
+modeling_gemma.GemmaModel.__init__ = patched_gemma_model_init
+def patched_gemma_pretrained_model_init_weights(self, module: nn.Module):
+    """Initializes the weights of the GemmaPreTrainedModel.
+    Args:
+        self: The GemmaPreTrainedModel instance.
+        module: The module to initialize.
+    """
+    std = self.config.initializer_range
+    if isinstance(module, nn.Linear):
+        module.weight.data.normal_(mean=0.0, std=std)
+        if module.bias is not None:
+            module.bias.data.zero_()
+    elif isinstance(module, nn.Embedding):
+        module.weight.data.normal_(mean=0.0, std=std)
+        if module.padding_idx is not None:
+            module.weight.data[module.padding_idx].zero_()
+    elif isinstance(module, modeling_gemma.GemmaRMSNorm):
+        if hasattr(module, "weight"):
+            module.weight.data.fill_(1.0)
+modeling_gemma.GemmaPreTrainedModel._init_weights = patched_gemma_pretrained_model_init_weights
+def patched_paligemma_model_get_image_features(self, pixel_values: torch.FloatTensor) -> torch.Tensor:
+    """Obtains image last hidden states from the vision tower and apply multimodal projection.
+    Args:
+        self: The PaliGemmaModel instance.
+        pixel_values: The tensors corresponding to the input images.
+            Shape: (batch_size, channels, height, width).
+    Returns:
+        Image feature tensor of shape (num_images, image_length, embed_dim).
+    """
+    image_outputs = self.vision_tower(pixel_values)
+    selected_image_feature = image_outputs.last_hidden_state
+    image_features = self.multi_modal_projector(selected_image_feature)
+    return image_features
+PaliGemmaModel.get_image_features = patched_paligemma_model_get_image_features

opentau 0.1.0__tar.gz → 0.1.2__tar.gz

opentau 0.1.0tar.gz → 0.1.2tar.gz