PyPI - dgenerate-ultralytics-headless - Versions diffs - 8.3.137__py3-none-any.whl → 8.3.224__py3-none-any.whl - Mend

dgenerate-ultralytics-headless 8.3.137py3-none-any.whl → 8.3.224py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (215) hide show

{dgenerate_ultralytics_headless-8.3.137.dist-info → dgenerate_ultralytics_headless-8.3.224.dist-info}/METADATA +41 -34
dgenerate_ultralytics_headless-8.3.224.dist-info/RECORD +285 -0
{dgenerate_ultralytics_headless-8.3.137.dist-info → dgenerate_ultralytics_headless-8.3.224.dist-info}/WHEEL +1 -1
tests/__init__.py +7 -6
tests/conftest.py +15 -39
tests/test_cli.py +17 -17
tests/test_cuda.py +17 -8
tests/test_engine.py +36 -10
tests/test_exports.py +98 -37
tests/test_integrations.py +12 -15
tests/test_python.py +126 -82
tests/test_solutions.py +319 -135
ultralytics/__init__.py +27 -9
ultralytics/cfg/__init__.py +83 -87
ultralytics/cfg/datasets/Argoverse.yaml +4 -4
ultralytics/cfg/datasets/DOTAv1.5.yaml +2 -2
ultralytics/cfg/datasets/DOTAv1.yaml +2 -2
ultralytics/cfg/datasets/GlobalWheat2020.yaml +2 -2
ultralytics/cfg/datasets/HomeObjects-3K.yaml +4 -5
ultralytics/cfg/datasets/ImageNet.yaml +3 -3
ultralytics/cfg/datasets/Objects365.yaml +24 -20
ultralytics/cfg/datasets/SKU-110K.yaml +9 -9
ultralytics/cfg/datasets/VOC.yaml +10 -13
ultralytics/cfg/datasets/VisDrone.yaml +43 -33
ultralytics/cfg/datasets/african-wildlife.yaml +5 -5
ultralytics/cfg/datasets/brain-tumor.yaml +4 -5
ultralytics/cfg/datasets/carparts-seg.yaml +5 -5
ultralytics/cfg/datasets/coco-pose.yaml +26 -4
ultralytics/cfg/datasets/coco.yaml +4 -4
ultralytics/cfg/datasets/coco128-seg.yaml +2 -2
ultralytics/cfg/datasets/coco128.yaml +2 -2
ultralytics/cfg/datasets/coco8-grayscale.yaml +103 -0
ultralytics/cfg/datasets/coco8-multispectral.yaml +2 -2
ultralytics/cfg/datasets/coco8-pose.yaml +23 -2
ultralytics/cfg/datasets/coco8-seg.yaml +2 -2
ultralytics/cfg/datasets/coco8.yaml +2 -2
ultralytics/cfg/datasets/construction-ppe.yaml +32 -0
ultralytics/cfg/datasets/crack-seg.yaml +5 -5
ultralytics/cfg/datasets/dog-pose.yaml +32 -4
ultralytics/cfg/datasets/dota8-multispectral.yaml +2 -2
ultralytics/cfg/datasets/dota8.yaml +2 -2
ultralytics/cfg/datasets/hand-keypoints.yaml +29 -4
ultralytics/cfg/datasets/lvis.yaml +9 -9
ultralytics/cfg/datasets/medical-pills.yaml +4 -5
ultralytics/cfg/datasets/open-images-v7.yaml +7 -10
ultralytics/cfg/datasets/package-seg.yaml +5 -5
ultralytics/cfg/datasets/signature.yaml +4 -4
ultralytics/cfg/datasets/tiger-pose.yaml +20 -4
ultralytics/cfg/datasets/xView.yaml +5 -5
ultralytics/cfg/default.yaml +96 -93
ultralytics/cfg/trackers/botsort.yaml +16 -17
ultralytics/cfg/trackers/bytetrack.yaml +9 -11
ultralytics/data/__init__.py +4 -4
ultralytics/data/annotator.py +12 -12
ultralytics/data/augment.py +531 -564
ultralytics/data/base.py +76 -81
ultralytics/data/build.py +206 -42
ultralytics/data/converter.py +179 -78
ultralytics/data/dataset.py +121 -121
ultralytics/data/loaders.py +114 -91
ultralytics/data/split.py +28 -15
ultralytics/data/split_dota.py +67 -48
ultralytics/data/utils.py +110 -89
ultralytics/engine/exporter.py +422 -460
ultralytics/engine/model.py +224 -252
ultralytics/engine/predictor.py +94 -89
ultralytics/engine/results.py +345 -595
ultralytics/engine/trainer.py +231 -134
ultralytics/engine/tuner.py +279 -73
ultralytics/engine/validator.py +53 -46
ultralytics/hub/__init__.py +26 -28
ultralytics/hub/auth.py +30 -16
ultralytics/hub/google/__init__.py +34 -36
ultralytics/hub/session.py +53 -77
ultralytics/hub/utils.py +23 -109
ultralytics/models/__init__.py +1 -1
ultralytics/models/fastsam/__init__.py +1 -1
ultralytics/models/fastsam/model.py +36 -18
ultralytics/models/fastsam/predict.py +33 -44
ultralytics/models/fastsam/utils.py +4 -5
ultralytics/models/fastsam/val.py +12 -14
ultralytics/models/nas/__init__.py +1 -1
ultralytics/models/nas/model.py +16 -20
ultralytics/models/nas/predict.py +12 -14
ultralytics/models/nas/val.py +4 -5
ultralytics/models/rtdetr/__init__.py +1 -1
ultralytics/models/rtdetr/model.py +9 -9
ultralytics/models/rtdetr/predict.py +22 -17
ultralytics/models/rtdetr/train.py +20 -16
ultralytics/models/rtdetr/val.py +79 -59
ultralytics/models/sam/__init__.py +8 -2
ultralytics/models/sam/amg.py +53 -38
ultralytics/models/sam/build.py +29 -31
ultralytics/models/sam/model.py +33 -38
ultralytics/models/sam/modules/blocks.py +159 -182
ultralytics/models/sam/modules/decoders.py +38 -47
ultralytics/models/sam/modules/encoders.py +114 -133
ultralytics/models/sam/modules/memory_attention.py +38 -31
ultralytics/models/sam/modules/sam.py +114 -93
ultralytics/models/sam/modules/tiny_encoder.py +268 -291
ultralytics/models/sam/modules/transformer.py +59 -66
ultralytics/models/sam/modules/utils.py +55 -72
ultralytics/models/sam/predict.py +745 -341
ultralytics/models/utils/loss.py +118 -107
ultralytics/models/utils/ops.py +118 -71
ultralytics/models/yolo/__init__.py +1 -1
ultralytics/models/yolo/classify/predict.py +28 -26
ultralytics/models/yolo/classify/train.py +50 -81
ultralytics/models/yolo/classify/val.py +68 -61
ultralytics/models/yolo/detect/predict.py +12 -15
ultralytics/models/yolo/detect/train.py +56 -46
ultralytics/models/yolo/detect/val.py +279 -223
ultralytics/models/yolo/model.py +167 -86
ultralytics/models/yolo/obb/predict.py +7 -11
ultralytics/models/yolo/obb/train.py +23 -25
ultralytics/models/yolo/obb/val.py +107 -99
ultralytics/models/yolo/pose/__init__.py +1 -1
ultralytics/models/yolo/pose/predict.py +12 -14
ultralytics/models/yolo/pose/train.py +31 -69
ultralytics/models/yolo/pose/val.py +119 -254
ultralytics/models/yolo/segment/predict.py +21 -25
ultralytics/models/yolo/segment/train.py +12 -66
ultralytics/models/yolo/segment/val.py +126 -305
ultralytics/models/yolo/world/train.py +53 -45
ultralytics/models/yolo/world/train_world.py +51 -32
ultralytics/models/yolo/yoloe/__init__.py +7 -7
ultralytics/models/yolo/yoloe/predict.py +30 -37
ultralytics/models/yolo/yoloe/train.py +89 -71
ultralytics/models/yolo/yoloe/train_seg.py +15 -17
ultralytics/models/yolo/yoloe/val.py +56 -41
ultralytics/nn/__init__.py +9 -11
ultralytics/nn/autobackend.py +179 -107
ultralytics/nn/modules/__init__.py +67 -67
ultralytics/nn/modules/activation.py +8 -7
ultralytics/nn/modules/block.py +302 -323
ultralytics/nn/modules/conv.py +61 -104
ultralytics/nn/modules/head.py +488 -186
ultralytics/nn/modules/transformer.py +183 -123
ultralytics/nn/modules/utils.py +15 -20
ultralytics/nn/tasks.py +327 -203
ultralytics/nn/text_model.py +81 -65
ultralytics/py.typed +1 -0
ultralytics/solutions/__init__.py +12 -12
ultralytics/solutions/ai_gym.py +19 -27
ultralytics/solutions/analytics.py +36 -26
ultralytics/solutions/config.py +29 -28
ultralytics/solutions/distance_calculation.py +23 -24
ultralytics/solutions/heatmap.py +17 -19
ultralytics/solutions/instance_segmentation.py +21 -19
ultralytics/solutions/object_blurrer.py +16 -17
ultralytics/solutions/object_counter.py +48 -53
ultralytics/solutions/object_cropper.py +22 -16
ultralytics/solutions/parking_management.py +61 -58
ultralytics/solutions/queue_management.py +19 -19
ultralytics/solutions/region_counter.py +63 -50
ultralytics/solutions/security_alarm.py +22 -25
ultralytics/solutions/similarity_search.py +107 -60
ultralytics/solutions/solutions.py +343 -262
ultralytics/solutions/speed_estimation.py +35 -31
ultralytics/solutions/streamlit_inference.py +104 -40
ultralytics/solutions/templates/similarity-search.html +31 -24
ultralytics/solutions/trackzone.py +24 -24
ultralytics/solutions/vision_eye.py +11 -12
ultralytics/trackers/__init__.py +1 -1
ultralytics/trackers/basetrack.py +18 -27
ultralytics/trackers/bot_sort.py +48 -39
ultralytics/trackers/byte_tracker.py +94 -94
ultralytics/trackers/track.py +7 -16
ultralytics/trackers/utils/gmc.py +37 -69
ultralytics/trackers/utils/kalman_filter.py +68 -76
ultralytics/trackers/utils/matching.py +13 -17
ultralytics/utils/__init__.py +251 -275
ultralytics/utils/autobatch.py +19 -7
ultralytics/utils/autodevice.py +68 -38
ultralytics/utils/benchmarks.py +169 -130
ultralytics/utils/callbacks/base.py +12 -13
ultralytics/utils/callbacks/clearml.py +14 -15
ultralytics/utils/callbacks/comet.py +139 -66
ultralytics/utils/callbacks/dvc.py +19 -27
ultralytics/utils/callbacks/hub.py +8 -6
ultralytics/utils/callbacks/mlflow.py +6 -10
ultralytics/utils/callbacks/neptune.py +11 -19
ultralytics/utils/callbacks/platform.py +73 -0
ultralytics/utils/callbacks/raytune.py +3 -4
ultralytics/utils/callbacks/tensorboard.py +9 -12
ultralytics/utils/callbacks/wb.py +33 -30
ultralytics/utils/checks.py +163 -114
ultralytics/utils/cpu.py +89 -0
ultralytics/utils/dist.py +24 -20
ultralytics/utils/downloads.py +176 -146
ultralytics/utils/errors.py +11 -13
ultralytics/utils/events.py +113 -0
ultralytics/utils/export/__init__.py +7 -0
ultralytics/utils/{export.py → export/engine.py} +81 -63
ultralytics/utils/export/imx.py +294 -0
ultralytics/utils/export/tensorflow.py +217 -0
ultralytics/utils/files.py +33 -36
ultralytics/utils/git.py +137 -0
ultralytics/utils/instance.py +105 -120
ultralytics/utils/logger.py +404 -0
ultralytics/utils/loss.py +99 -61
ultralytics/utils/metrics.py +649 -478
ultralytics/utils/nms.py +337 -0
ultralytics/utils/ops.py +263 -451
ultralytics/utils/patches.py +70 -31
ultralytics/utils/plotting.py +253 -223
ultralytics/utils/tal.py +48 -61
ultralytics/utils/torch_utils.py +244 -251
ultralytics/utils/tqdm.py +438 -0
ultralytics/utils/triton.py +22 -23
ultralytics/utils/tuner.py +11 -10
dgenerate_ultralytics_headless-8.3.137.dist-info/RECORD +0 -272
{dgenerate_ultralytics_headless-8.3.137.dist-info → dgenerate_ultralytics_headless-8.3.224.dist-info}/entry_points.txt +0 -0
{dgenerate_ultralytics_headless-8.3.137.dist-info → dgenerate_ultralytics_headless-8.3.224.dist-info}/licenses/LICENSE +0 -0
{dgenerate_ultralytics_headless-8.3.137.dist-info → dgenerate_ultralytics_headless-8.3.224.dist-info}/top_level.txt +0 -0

ultralytics/models/utils/ops.py CHANGED Viewed

@@ -1,5 +1,9 @@
 # Ultralytics 🚀 AGPL-3.0 License - https://ultralytics.com/license
+from __future__ import annotations
+from typing import Any
 import torch
 import torch.nn as nn
 import torch.nn.functional as F
@@ -10,41 +14,57 @@ from ultralytics.utils.ops import xywh2xyxy, xyxy2xywh
 class HungarianMatcher(nn.Module):
-    """
-    A module implementing the HungarianMatcher, which is a differentiable module to solve the assignment problem in an
-    end-to-end fashion.
+    """A module implementing the HungarianMatcher for optimal assignment between predictions and ground truth.
-    HungarianMatcher performs optimal assignment over the predicted and ground truth bounding boxes using a cost
-    function that considers classification scores, bounding box coordinates, and optionally, mask predictions.
+    HungarianMatcher performs optimal bipartite assignment over predicted and ground truth bounding boxes using a cost
+    function that considers classification scores, bounding box coordinates, and optionally mask predictions. This is
+    used in end-to-end object detection models like DETR.
     Attributes:
-        cost_gain (dict): Dictionary of cost coefficients: 'class', 'bbox', 'giou', 'mask', and 'dice'.
-        use_fl (bool): Indicates whether to use Focal Loss for the classification cost calculation.
-        with_mask (bool): Indicates whether the model makes mask predictions.
-        num_sample_points (int): The number of sample points used in mask cost calculation.
-        alpha (float): The alpha factor in Focal Loss calculation.
-        gamma (float): The gamma factor in Focal Loss calculation.
+        cost_gain (dict[str, float]): Dictionary of cost coefficients for 'class', 'bbox', 'giou', 'mask', and 'dice'
+            components.
+        use_fl (bool): Whether to use Focal Loss for classification cost calculation.
+        with_mask (bool): Whether the model makes mask predictions.
+        num_sample_points (int): Number of sample points used in mask cost calculation.
+        alpha (float): Alpha factor in Focal Loss calculation.
+        gamma (float): Gamma factor in Focal Loss calculation.
     Methods:
-        forward: Computes the assignment between predictions and ground truths for a batch.
-        _cost_mask: Computes the mask cost and dice cost if masks are predicted.
+        forward: Compute optimal assignment between predictions and ground truths for a batch.
+        _cost_mask: Compute mask cost and dice cost if masks are predicted.
+    Examples:
+        Initialize a HungarianMatcher with custom cost gains
+        >>> matcher = HungarianMatcher(cost_gain={"class": 2, "bbox": 5, "giou": 2})
+        Perform matching between predictions and ground truth
+        >>> pred_boxes = torch.rand(2, 100, 4)  # batch_size=2, num_queries=100
+        >>> pred_scores = torch.rand(2, 100, 80)  # 80 classes
+        >>> gt_boxes = torch.rand(10, 4)  # 10 ground truth boxes
+        >>> gt_classes = torch.randint(0, 80, (10,))
+        >>> gt_groups = [5, 5]  # 5 GT boxes per image
+        >>> indices = matcher(pred_boxes, pred_scores, gt_boxes, gt_classes, gt_groups)
     """
-    def __init__(self, cost_gain=None, use_fl=True, with_mask=False, num_sample_points=12544, alpha=0.25, gamma=2.0):
-        """
-        Initialize a HungarianMatcher module for optimal assignment of predicted and ground truth bounding boxes.
-        The HungarianMatcher uses a cost function that considers classification scores, bounding box coordinates,
-        and optionally mask predictions to perform optimal bipartite matching between predictions and ground truths.
+    def __init__(
+        self,
+        cost_gain: dict[str, float] | None = None,
+        use_fl: bool = True,
+        with_mask: bool = False,
+        num_sample_points: int = 12544,
+        alpha: float = 0.25,
+        gamma: float = 2.0,
+    ):
+        """Initialize HungarianMatcher for optimal assignment of predicted and ground truth bounding boxes.
         Args:
-            cost_gain (dict, optional): Dictionary of cost coefficients for different components of the matching cost.
-                Should contain keys 'class', 'bbox', 'giou', 'mask', and 'dice'.
-            use_fl (bool, optional): Whether to use Focal Loss for the classification cost calculation.
-            with_mask (bool, optional): Whether the model makes mask predictions.
-            num_sample_points (int, optional): Number of sample points used in mask cost calculation.
-            alpha (float, optional): Alpha factor in Focal Loss calculation.
-            gamma (float, optional): Gamma factor in Focal Loss calculation.
+            cost_gain (dict[str, float], optional): Dictionary of cost coefficients for different matching cost
+                components. Should contain keys 'class', 'bbox', 'giou', 'mask', and 'dice'.
+            use_fl (bool): Whether to use Focal Loss for classification cost calculation.
+            with_mask (bool): Whether the model makes mask predictions.
+            num_sample_points (int): Number of sample points used in mask cost calculation.
+            alpha (float): Alpha factor in Focal Loss calculation.
+            gamma (float): Gamma factor in Focal Loss calculation.
         """
         super().__init__()
         if cost_gain is None:
@@ -56,41 +76,48 @@ class HungarianMatcher(nn.Module):
         self.alpha = alpha
         self.gamma = gamma
-    def forward(self, pred_bboxes, pred_scores, gt_bboxes, gt_cls, gt_groups, masks=None, gt_mask=None):
-        """
-        Forward pass for HungarianMatcher. Computes costs based on prediction and ground truth and finds the optimal
-        matching between predictions and ground truth based on these costs.
+    def forward(
+        self,
+        pred_bboxes: torch.Tensor,
+        pred_scores: torch.Tensor,
+        gt_bboxes: torch.Tensor,
+        gt_cls: torch.Tensor,
+        gt_groups: list[int],
+        masks: torch.Tensor | None = None,
+        gt_mask: list[torch.Tensor] | None = None,
+    ) -> list[tuple[torch.Tensor, torch.Tensor]]:
+        """Compute optimal assignment between predictions and ground truth using Hungarian algorithm.
+        This method calculates matching costs based on classification scores, bounding box coordinates, and optionally
+        mask predictions, then finds the optimal bipartite assignment between predictions and ground truth.
         Args:
             pred_bboxes (torch.Tensor): Predicted bounding boxes with shape (batch_size, num_queries, 4).
-            pred_scores (torch.Tensor): Predicted scores with shape (batch_size, num_queries, num_classes).
-            gt_cls (torch.Tensor): Ground truth classes with shape (num_gts, ).
+            pred_scores (torch.Tensor): Predicted classification scores with shape (batch_size, num_queries,
+                num_classes).
             gt_bboxes (torch.Tensor): Ground truth bounding boxes with shape (num_gts, 4).
-            gt_groups (List[int]): List of length equal to batch size, containing the number of ground truths for
-                each image.
+            gt_cls (torch.Tensor): Ground truth class labels with shape (num_gts,).
+            gt_groups (list[int]): Number of ground truth boxes for each image in the batch.
             masks (torch.Tensor, optional): Predicted masks with shape (batch_size, num_queries, height, width).
-            gt_mask (List[torch.Tensor], optional): List of ground truth masks, each with shape (num_masks, Height, Width).
+            gt_mask (list[torch.Tensor], optional): Ground truth masks, each with shape (num_masks, Height, Width).
         Returns:
-            (List[Tuple[torch.Tensor, torch.Tensor]]): A list of size batch_size, each element is a tuple (index_i, index_j), where:
-                - index_i is the tensor of indices of the selected predictions (in order)
-                - index_j is the tensor of indices of the corresponding selected ground truth targets (in order)
-                For each batch element, it holds:
-                    len(index_i) = len(index_j) = min(num_queries, num_target_boxes)
+            (list[tuple[torch.Tensor, torch.Tensor]]): A list of size batch_size, each element is a tuple (index_i,
+                index_j), where index_i is the tensor of indices of the selected predictions (in order) and index_j is
+                the tensor of indices of the corresponding selected ground truth targets (in order).
+            For each batch element, it holds: len(index_i) = len(index_j) = min(num_queries, num_target_boxes).
         """
         bs, nq, nc = pred_scores.shape
         if sum(gt_groups) == 0:
             return [(torch.tensor([], dtype=torch.long), torch.tensor([], dtype=torch.long)) for _ in range(bs)]
-        # We flatten to compute the cost matrices in a batch
-        # (batch_size * num_queries, num_classes)
+        # Flatten to compute cost matrices in batch format
         pred_scores = pred_scores.detach().view(-1, nc)
         pred_scores = F.sigmoid(pred_scores) if self.use_fl else F.softmax(pred_scores, dim=-1)
-        # (batch_size * num_queries, 4)
         pred_bboxes = pred_bboxes.detach().view(-1, 4)
-        # Compute the classification cost
+        # Compute classification cost
         pred_scores = pred_scores[:, gt_cls]
         if self.use_fl:
             neg_cost_class = (1 - self.alpha) * (pred_scores**self.gamma) * (-(1 - pred_scores + 1e-8).log())
@@ -99,23 +126,24 @@ class HungarianMatcher(nn.Module):
         else:
             cost_class = -pred_scores
-        # Compute the L1 cost between boxes
+        # Compute L1 cost between boxes
         cost_bbox = (pred_bboxes.unsqueeze(1) - gt_bboxes.unsqueeze(0)).abs().sum(-1)  # (bs*num_queries, num_gt)
-        # Compute the GIoU cost between boxes, (bs*num_queries, num_gt)
+        # Compute GIoU cost between boxes, (bs*num_queries, num_gt)
         cost_giou = 1.0 - bbox_iou(pred_bboxes.unsqueeze(1), gt_bboxes.unsqueeze(0), xywh=True, GIoU=True).squeeze(-1)
-        # Final cost matrix
+        # Combine costs into final cost matrix
         C = (
             self.cost_gain["class"] * cost_class
             + self.cost_gain["bbox"] * cost_bbox
             + self.cost_gain["giou"] * cost_giou
         )
-        # Compute the mask cost and dice cost
+        # Add mask costs if available
         if self.with_mask:
             C += self._cost_mask(bs, gt_groups, masks, gt_mask)
-        # Set invalid values (NaNs and infinities) to 0 (fixes ValueError: matrix contains invalid numeric entries)
+        # Set invalid values (NaNs and infinities) to 0
         C[C.isnan() | C.isinf()] = 0.0
         C = C.view(bs, nq, -1).cpu()
@@ -158,28 +186,48 @@ class HungarianMatcher(nn.Module):
 def get_cdn_group(
-    batch, num_classes, num_queries, class_embed, num_dn=100, cls_noise_ratio=0.5, box_noise_scale=1.0, training=False
-):
-    """
-    Get contrastive denoising training group with positive and negative samples from ground truths.
+    batch: dict[str, Any],
+    num_classes: int,
+    num_queries: int,
+    class_embed: torch.Tensor,
+    num_dn: int = 100,
+    cls_noise_ratio: float = 0.5,
+    box_noise_scale: float = 1.0,
+    training: bool = False,
+) -> tuple[torch.Tensor | None, torch.Tensor | None, torch.Tensor | None, dict[str, Any] | None]:
+    """Generate contrastive denoising training group with positive and negative samples from ground truths.
+    This function creates denoising queries for contrastive denoising training by adding noise to ground truth bounding
+    boxes and class labels. It generates both positive and negative samples to improve model robustness.
     Args:
-        batch (dict): A dict that includes 'gt_cls' (torch.Tensor with shape (num_gts, )), 'gt_bboxes'
-            (torch.Tensor with shape (num_gts, 4)), 'gt_groups' (List[int]) which is a list of batch size length
-            indicating the number of gts of each image.
-        num_classes (int): Number of classes.
-        num_queries (int): Number of queries.
-        class_embed (torch.Tensor): Embedding weights to map class labels to embedding space.
-        num_dn (int, optional): Number of denoising queries.
-        cls_noise_ratio (float, optional): Noise ratio for class labels.
-        box_noise_scale (float, optional): Noise scale for bounding box coordinates.
-        training (bool, optional): If it's in training mode.
+        batch (dict[str, Any]): Batch dictionary containing 'gt_cls' (torch.Tensor with shape (num_gts,)), 'gt_bboxes'
+            (torch.Tensor with shape (num_gts, 4)), and 'gt_groups' (list[int]) indicating number of ground truths
+            per image.
+        num_classes (int): Total number of object classes.
+        num_queries (int): Number of object queries.
+        class_embed (torch.Tensor): Class embedding weights to map labels to embedding space.
+        num_dn (int): Number of denoising queries to generate.
+        cls_noise_ratio (float): Noise ratio for class labels.
+        box_noise_scale (float): Noise scale for bounding box coordinates.
+        training (bool): Whether model is in training mode.
     Returns:
-        padding_cls (Optional[torch.Tensor]): The modified class embeddings for denoising.
-        padding_bbox (Optional[torch.Tensor]): The modified bounding boxes for denoising.
-        attn_mask (Optional[torch.Tensor]): The attention mask for denoising.
-        dn_meta (Optional[Dict]): Meta information for denoising.
+        padding_cls (torch.Tensor | None): Modified class embeddings for denoising with shape (bs, num_dn, embed_dim).
+        padding_bbox (torch.Tensor | None): Modified bounding boxes for denoising with shape (bs, num_dn, 4).
+        attn_mask (torch.Tensor | None): Attention mask for denoising with shape (tgt_size, tgt_size).
+        dn_meta (dict[str, Any] | None): Meta information dictionary containing denoising parameters.
+    Examples:
+        Generate denoising group for training
+        >>> batch = {
+        ...     "cls": torch.tensor([0, 1, 2]),
+        ...     "bboxes": torch.rand(3, 4),
+        ...     "batch_idx": torch.tensor([0, 0, 1]),
+        ...     "gt_groups": [2, 1],
+        ... }
+        >>> class_embed = torch.rand(80, 256)  # 80 classes, 256 embedding dim
+        >>> cdn_outputs = get_cdn_group(batch, 80, 100, class_embed, training=True)
     """
     if (not training) or num_dn <= 0 or batch is None:
         return None, None, None, None
@@ -197,7 +245,7 @@ def get_cdn_group(
     gt_bbox = batch["bboxes"]  # bs*num, 4
     b_idx = batch["batch_idx"]
-    # Each group has positive and negative queries.
+    # Each group has positive and negative queries
     dn_cls = gt_cls.repeat(2 * num_group)  # (2*num_group*bs*num, )
     dn_bbox = gt_bbox.repeat(2 * num_group, 1)  # 2*num_group*bs*num, 4
     dn_b_idx = b_idx.repeat(2 * num_group).view(-1)  # (2*num_group*bs*num, )
@@ -207,10 +255,10 @@ def get_cdn_group(
     neg_idx = torch.arange(total_num * num_group, dtype=torch.long, device=gt_bbox.device) + num_group * total_num
     if cls_noise_ratio > 0:
-        # Half of bbox prob
+        # Apply class label noise to half of the samples
         mask = torch.rand(dn_cls.shape) < (cls_noise_ratio * 0.5)
         idx = torch.nonzero(mask).squeeze(-1)
-        # Randomly put a new one here
+        # Randomly assign new class labels
         new_label = torch.randint_like(idx, 0, num_classes, dtype=dn_cls.dtype, device=dn_cls.device)
         dn_cls[idx] = new_label
@@ -229,7 +277,6 @@ def get_cdn_group(
         dn_bbox = torch.logit(dn_bbox, eps=1e-6)  # inverse sigmoid
     num_dn = int(max_nums * 2 * num_group)  # total denoising queries
-    # class_embed = torch.cat([class_embed, torch.zeros([1, class_embed.shape[-1]], device=class_embed.device)])
     dn_cls_embed = class_embed[dn_cls]  # bs*num * 2 * num_group, 256
     padding_cls = torch.zeros(bs, num_dn, dn_cls_embed.shape[-1], device=gt_cls.device)
     padding_bbox = torch.zeros(bs, num_dn, 4, device=gt_bbox.device)

ultralytics/models/yolo/__init__.py CHANGED Viewed

@@ -4,4 +4,4 @@ from ultralytics.models.yolo import classify, detect, obb, pose, segment, world,
 from .model import YOLO, YOLOE, YOLOWorld
-__all__ = "classify", "segment", "detect", "pose", "obb", "world", "yoloe", "YOLO", "YOLOWorld", "YOLOE"
+__all__ = "YOLO", "YOLOE", "YOLOWorld", "classify", "detect", "obb", "pose", "segment", "world", "yoloe"

ultralytics/models/yolo/classify/predict.py CHANGED Viewed

@@ -4,81 +4,83 @@ import cv2
 import torch
 from PIL import Image
+from ultralytics.data.augment import classify_transforms
 from ultralytics.engine.predictor import BasePredictor
 from ultralytics.engine.results import Results
 from ultralytics.utils import DEFAULT_CFG, ops
 class ClassificationPredictor(BasePredictor):
-    """
-    A class extending the BasePredictor class for prediction based on a classification model.
+    """A class extending the BasePredictor class for prediction based on a classification model.
-    This predictor handles the specific requirements of classification models, including preprocessing images
-    and postprocessing predictions to generate classification results.
+    This predictor handles the specific requirements of classification models, including preprocessing images and
+    postprocessing predictions to generate classification results.
     Attributes:
         args (dict): Configuration arguments for the predictor.
-        _legacy_transform_name (str): Name of the legacy transform class for backward compatibility.
     Methods:
         preprocess: Convert input images to model-compatible format.
         postprocess: Process model predictions into Results objects.
-    Notes:
-        - Torchvision classification models can also be passed to the 'model' argument, i.e. model='resnet18'.
     Examples:
         >>> from ultralytics.utils import ASSETS
         >>> from ultralytics.models.yolo.classify import ClassificationPredictor
         >>> args = dict(model="yolo11n-cls.pt", source=ASSETS)
         >>> predictor = ClassificationPredictor(overrides=args)
         >>> predictor.predict_cli()
+    Notes:
+        - Torchvision classification models can also be passed to the 'model' argument, i.e. model='resnet18'.
     """
     def __init__(self, cfg=DEFAULT_CFG, overrides=None, _callbacks=None):
-        """
-        Initialize the ClassificationPredictor with the specified configuration and set task to 'classify'.
+        """Initialize the ClassificationPredictor with the specified configuration and set task to 'classify'.
         This constructor initializes a ClassificationPredictor instance, which extends BasePredictor for classification
         tasks. It ensures the task is set to 'classify' regardless of input configuration.
         Args:
-            cfg (dict): Default configuration dictionary containing prediction settings. Defaults to DEFAULT_CFG.
+            cfg (dict): Default configuration dictionary containing prediction settings.
             overrides (dict, optional): Configuration overrides that take precedence over cfg.
             _callbacks (list, optional): List of callback functions to be executed during prediction.
         """
         super().__init__(cfg, overrides, _callbacks)
         self.args.task = "classify"
-        self._legacy_transform_name = "ultralytics.yolo.data.augment.ToTensor"
+    def setup_source(self, source):
+        """Set up source and inference mode and classify transforms."""
+        super().setup_source(source)
+        updated = (
+            self.model.model.transforms.transforms[0].size != max(self.imgsz)
+            if hasattr(self.model.model, "transforms") and hasattr(self.model.model.transforms.transforms[0], "size")
+            else False
+        )
+        self.transforms = (
+            classify_transforms(self.imgsz) if updated or not self.model.pt else self.model.model.transforms
+        )
     def preprocess(self, img):
         """Convert input images to model-compatible tensor format with appropriate normalization."""
         if not isinstance(img, torch.Tensor):
-            is_legacy_transform = any(
-                self._legacy_transform_name in str(transform) for transform in self.transforms.transforms
+            img = torch.stack(
+                [self.transforms(Image.fromarray(cv2.cvtColor(im, cv2.COLOR_BGR2RGB))) for im in img], dim=0
             )
-            if is_legacy_transform:  # to handle legacy transforms
-                img = torch.stack([self.transforms(im) for im in img], dim=0)
-            else:
-                img = torch.stack(
-                    [self.transforms(Image.fromarray(cv2.cvtColor(im, cv2.COLOR_BGR2RGB))) for im in img], dim=0
-                )
         img = (img if isinstance(img, torch.Tensor) else torch.from_numpy(img)).to(self.model.device)
-        return img.half() if self.model.fp16 else img.float()  # uint8 to fp16/32
+        return img.half() if self.model.fp16 else img.float()  # Convert uint8 to fp16/32
     def postprocess(self, preds, img, orig_imgs):
-        """
-        Process predictions to return Results objects with classification probabilities.
+        """Process predictions to return Results objects with classification probabilities.
         Args:
             preds (torch.Tensor): Raw predictions from the model.
             img (torch.Tensor): Input images after preprocessing.
-            orig_imgs (List[np.ndarray] | torch.Tensor): Original images before preprocessing.
+            orig_imgs (list[np.ndarray] | torch.Tensor): Original images before preprocessing.
         Returns:
-            (List[Results]): List of Results objects containing classification results for each image.
+            (list[Results]): List of Results objects containing classification results for each image.
         """
-        if not isinstance(orig_imgs, list):  # input images are a torch.Tensor, not a list
+        if not isinstance(orig_imgs, list):  # Input images are a torch.Tensor, not a list
             orig_imgs = ops.convert_torch2numpy_batch(orig_imgs)
         preds = preds[0] if isinstance(preds, (list, tuple)) else preds

dgenerate-ultralytics-headless 8.3.137__py3-none-any.whl → 8.3.224__py3-none-any.whl

dgenerate-ultralytics-headless 8.3.137py3-none-any.whl → 8.3.224py3-none-any.whl