PyPI - dgenerate-ultralytics-headless - Versions diffs - 8.3.134__py3-none-any.whl - Mend

dgenerate-ultralytics-headless 8.3.134__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (272) hide show

dgenerate_ultralytics_headless-8.3.134.dist-info/METADATA +400 -0
dgenerate_ultralytics_headless-8.3.134.dist-info/RECORD +272 -0
dgenerate_ultralytics_headless-8.3.134.dist-info/WHEEL +5 -0
dgenerate_ultralytics_headless-8.3.134.dist-info/entry_points.txt +3 -0
dgenerate_ultralytics_headless-8.3.134.dist-info/licenses/LICENSE +661 -0
dgenerate_ultralytics_headless-8.3.134.dist-info/top_level.txt +1 -0
tests/__init__.py +22 -0
tests/conftest.py +83 -0
tests/test_cli.py +138 -0
tests/test_cuda.py +215 -0
tests/test_engine.py +131 -0
tests/test_exports.py +236 -0
tests/test_integrations.py +154 -0
tests/test_python.py +694 -0
tests/test_solutions.py +187 -0
ultralytics/__init__.py +30 -0
ultralytics/assets/bus.jpg +0 -0
ultralytics/assets/zidane.jpg +0 -0
ultralytics/cfg/__init__.py +1023 -0
ultralytics/cfg/datasets/Argoverse.yaml +77 -0
ultralytics/cfg/datasets/DOTAv1.5.yaml +37 -0
ultralytics/cfg/datasets/DOTAv1.yaml +36 -0
ultralytics/cfg/datasets/GlobalWheat2020.yaml +68 -0
ultralytics/cfg/datasets/HomeObjects-3K.yaml +33 -0
ultralytics/cfg/datasets/ImageNet.yaml +2025 -0
ultralytics/cfg/datasets/Objects365.yaml +443 -0
ultralytics/cfg/datasets/SKU-110K.yaml +58 -0
ultralytics/cfg/datasets/VOC.yaml +106 -0
ultralytics/cfg/datasets/VisDrone.yaml +77 -0
ultralytics/cfg/datasets/african-wildlife.yaml +25 -0
ultralytics/cfg/datasets/brain-tumor.yaml +23 -0
ultralytics/cfg/datasets/carparts-seg.yaml +44 -0
ultralytics/cfg/datasets/coco-pose.yaml +42 -0
ultralytics/cfg/datasets/coco.yaml +118 -0
ultralytics/cfg/datasets/coco128-seg.yaml +101 -0
ultralytics/cfg/datasets/coco128.yaml +101 -0
ultralytics/cfg/datasets/coco8-multispectral.yaml +104 -0
ultralytics/cfg/datasets/coco8-pose.yaml +26 -0
ultralytics/cfg/datasets/coco8-seg.yaml +101 -0
ultralytics/cfg/datasets/coco8.yaml +101 -0
ultralytics/cfg/datasets/crack-seg.yaml +22 -0
ultralytics/cfg/datasets/dog-pose.yaml +24 -0
ultralytics/cfg/datasets/dota8-multispectral.yaml +38 -0
ultralytics/cfg/datasets/dota8.yaml +35 -0
ultralytics/cfg/datasets/hand-keypoints.yaml +26 -0
ultralytics/cfg/datasets/lvis.yaml +1240 -0
ultralytics/cfg/datasets/medical-pills.yaml +22 -0
ultralytics/cfg/datasets/open-images-v7.yaml +666 -0
ultralytics/cfg/datasets/package-seg.yaml +22 -0
ultralytics/cfg/datasets/signature.yaml +21 -0
ultralytics/cfg/datasets/tiger-pose.yaml +25 -0
ultralytics/cfg/datasets/xView.yaml +155 -0
ultralytics/cfg/default.yaml +127 -0
ultralytics/cfg/models/11/yolo11-cls-resnet18.yaml +17 -0
ultralytics/cfg/models/11/yolo11-cls.yaml +33 -0
ultralytics/cfg/models/11/yolo11-obb.yaml +50 -0
ultralytics/cfg/models/11/yolo11-pose.yaml +51 -0
ultralytics/cfg/models/11/yolo11-seg.yaml +50 -0
ultralytics/cfg/models/11/yolo11.yaml +50 -0
ultralytics/cfg/models/11/yoloe-11-seg.yaml +48 -0
ultralytics/cfg/models/11/yoloe-11.yaml +48 -0
ultralytics/cfg/models/12/yolo12-cls.yaml +32 -0
ultralytics/cfg/models/12/yolo12-obb.yaml +48 -0
ultralytics/cfg/models/12/yolo12-pose.yaml +49 -0
ultralytics/cfg/models/12/yolo12-seg.yaml +48 -0
ultralytics/cfg/models/12/yolo12.yaml +48 -0
ultralytics/cfg/models/rt-detr/rtdetr-l.yaml +53 -0
ultralytics/cfg/models/rt-detr/rtdetr-resnet101.yaml +45 -0
ultralytics/cfg/models/rt-detr/rtdetr-resnet50.yaml +45 -0
ultralytics/cfg/models/rt-detr/rtdetr-x.yaml +57 -0
ultralytics/cfg/models/v10/yolov10b.yaml +45 -0
ultralytics/cfg/models/v10/yolov10l.yaml +45 -0
ultralytics/cfg/models/v10/yolov10m.yaml +45 -0
ultralytics/cfg/models/v10/yolov10n.yaml +45 -0
ultralytics/cfg/models/v10/yolov10s.yaml +45 -0
ultralytics/cfg/models/v10/yolov10x.yaml +45 -0
ultralytics/cfg/models/v3/yolov3-spp.yaml +49 -0
ultralytics/cfg/models/v3/yolov3-tiny.yaml +40 -0
ultralytics/cfg/models/v3/yolov3.yaml +49 -0
ultralytics/cfg/models/v5/yolov5-p6.yaml +62 -0
ultralytics/cfg/models/v5/yolov5.yaml +51 -0
ultralytics/cfg/models/v6/yolov6.yaml +56 -0
ultralytics/cfg/models/v8/yoloe-v8-seg.yaml +45 -0
ultralytics/cfg/models/v8/yoloe-v8.yaml +45 -0
ultralytics/cfg/models/v8/yolov8-cls-resnet101.yaml +28 -0
ultralytics/cfg/models/v8/yolov8-cls-resnet50.yaml +28 -0
ultralytics/cfg/models/v8/yolov8-cls.yaml +32 -0
ultralytics/cfg/models/v8/yolov8-ghost-p2.yaml +58 -0
ultralytics/cfg/models/v8/yolov8-ghost-p6.yaml +60 -0
ultralytics/cfg/models/v8/yolov8-ghost.yaml +50 -0
ultralytics/cfg/models/v8/yolov8-obb.yaml +49 -0
ultralytics/cfg/models/v8/yolov8-p2.yaml +57 -0
ultralytics/cfg/models/v8/yolov8-p6.yaml +59 -0
ultralytics/cfg/models/v8/yolov8-pose-p6.yaml +60 -0
ultralytics/cfg/models/v8/yolov8-pose.yaml +50 -0
ultralytics/cfg/models/v8/yolov8-rtdetr.yaml +49 -0
ultralytics/cfg/models/v8/yolov8-seg-p6.yaml +59 -0
ultralytics/cfg/models/v8/yolov8-seg.yaml +49 -0
ultralytics/cfg/models/v8/yolov8-world.yaml +51 -0
ultralytics/cfg/models/v8/yolov8-worldv2.yaml +49 -0
ultralytics/cfg/models/v8/yolov8.yaml +49 -0
ultralytics/cfg/models/v9/yolov9c-seg.yaml +41 -0
ultralytics/cfg/models/v9/yolov9c.yaml +41 -0
ultralytics/cfg/models/v9/yolov9e-seg.yaml +64 -0
ultralytics/cfg/models/v9/yolov9e.yaml +64 -0
ultralytics/cfg/models/v9/yolov9m.yaml +41 -0
ultralytics/cfg/models/v9/yolov9s.yaml +41 -0
ultralytics/cfg/models/v9/yolov9t.yaml +41 -0
ultralytics/cfg/trackers/botsort.yaml +22 -0
ultralytics/cfg/trackers/bytetrack.yaml +14 -0
ultralytics/data/__init__.py +26 -0
ultralytics/data/annotator.py +66 -0
ultralytics/data/augment.py +2945 -0
ultralytics/data/base.py +438 -0
ultralytics/data/build.py +258 -0
ultralytics/data/converter.py +754 -0
ultralytics/data/dataset.py +834 -0
ultralytics/data/loaders.py +676 -0
ultralytics/data/scripts/download_weights.sh +18 -0
ultralytics/data/scripts/get_coco.sh +61 -0
ultralytics/data/scripts/get_coco128.sh +18 -0
ultralytics/data/scripts/get_imagenet.sh +52 -0
ultralytics/data/split.py +125 -0
ultralytics/data/split_dota.py +325 -0
ultralytics/data/utils.py +777 -0
ultralytics/engine/__init__.py +1 -0
ultralytics/engine/exporter.py +1519 -0
ultralytics/engine/model.py +1156 -0
ultralytics/engine/predictor.py +502 -0
ultralytics/engine/results.py +1840 -0
ultralytics/engine/trainer.py +853 -0
ultralytics/engine/tuner.py +243 -0
ultralytics/engine/validator.py +377 -0
ultralytics/hub/__init__.py +168 -0
ultralytics/hub/auth.py +137 -0
ultralytics/hub/google/__init__.py +176 -0
ultralytics/hub/session.py +446 -0
ultralytics/hub/utils.py +248 -0
ultralytics/models/__init__.py +9 -0
ultralytics/models/fastsam/__init__.py +7 -0
ultralytics/models/fastsam/model.py +61 -0
ultralytics/models/fastsam/predict.py +181 -0
ultralytics/models/fastsam/utils.py +24 -0
ultralytics/models/fastsam/val.py +40 -0
ultralytics/models/nas/__init__.py +7 -0
ultralytics/models/nas/model.py +102 -0
ultralytics/models/nas/predict.py +58 -0
ultralytics/models/nas/val.py +39 -0
ultralytics/models/rtdetr/__init__.py +7 -0
ultralytics/models/rtdetr/model.py +63 -0
ultralytics/models/rtdetr/predict.py +84 -0
ultralytics/models/rtdetr/train.py +85 -0
ultralytics/models/rtdetr/val.py +191 -0
ultralytics/models/sam/__init__.py +6 -0
ultralytics/models/sam/amg.py +260 -0
ultralytics/models/sam/build.py +358 -0
ultralytics/models/sam/model.py +170 -0
ultralytics/models/sam/modules/__init__.py +1 -0
ultralytics/models/sam/modules/blocks.py +1129 -0
ultralytics/models/sam/modules/decoders.py +515 -0
ultralytics/models/sam/modules/encoders.py +854 -0
ultralytics/models/sam/modules/memory_attention.py +299 -0
ultralytics/models/sam/modules/sam.py +1006 -0
ultralytics/models/sam/modules/tiny_encoder.py +1002 -0
ultralytics/models/sam/modules/transformer.py +351 -0
ultralytics/models/sam/modules/utils.py +394 -0
ultralytics/models/sam/predict.py +1605 -0
ultralytics/models/utils/__init__.py +1 -0
ultralytics/models/utils/loss.py +455 -0
ultralytics/models/utils/ops.py +268 -0
ultralytics/models/yolo/__init__.py +7 -0
ultralytics/models/yolo/classify/__init__.py +7 -0
ultralytics/models/yolo/classify/predict.py +88 -0
ultralytics/models/yolo/classify/train.py +233 -0
ultralytics/models/yolo/classify/val.py +215 -0
ultralytics/models/yolo/detect/__init__.py +7 -0
ultralytics/models/yolo/detect/predict.py +124 -0
ultralytics/models/yolo/detect/train.py +217 -0
ultralytics/models/yolo/detect/val.py +451 -0
ultralytics/models/yolo/model.py +354 -0
ultralytics/models/yolo/obb/__init__.py +7 -0
ultralytics/models/yolo/obb/predict.py +66 -0
ultralytics/models/yolo/obb/train.py +81 -0
ultralytics/models/yolo/obb/val.py +283 -0
ultralytics/models/yolo/pose/__init__.py +7 -0
ultralytics/models/yolo/pose/predict.py +79 -0
ultralytics/models/yolo/pose/train.py +154 -0
ultralytics/models/yolo/pose/val.py +394 -0
ultralytics/models/yolo/segment/__init__.py +7 -0
ultralytics/models/yolo/segment/predict.py +113 -0
ultralytics/models/yolo/segment/train.py +123 -0
ultralytics/models/yolo/segment/val.py +428 -0
ultralytics/models/yolo/world/__init__.py +5 -0
ultralytics/models/yolo/world/train.py +119 -0
ultralytics/models/yolo/world/train_world.py +176 -0
ultralytics/models/yolo/yoloe/__init__.py +22 -0
ultralytics/models/yolo/yoloe/predict.py +169 -0
ultralytics/models/yolo/yoloe/train.py +298 -0
ultralytics/models/yolo/yoloe/train_seg.py +124 -0
ultralytics/models/yolo/yoloe/val.py +191 -0
ultralytics/nn/__init__.py +29 -0
ultralytics/nn/autobackend.py +842 -0
ultralytics/nn/modules/__init__.py +182 -0
ultralytics/nn/modules/activation.py +53 -0
ultralytics/nn/modules/block.py +1966 -0
ultralytics/nn/modules/conv.py +712 -0
ultralytics/nn/modules/head.py +880 -0
ultralytics/nn/modules/transformer.py +713 -0
ultralytics/nn/modules/utils.py +164 -0
ultralytics/nn/tasks.py +1627 -0
ultralytics/nn/text_model.py +351 -0
ultralytics/solutions/__init__.py +41 -0
ultralytics/solutions/ai_gym.py +116 -0
ultralytics/solutions/analytics.py +252 -0
ultralytics/solutions/config.py +106 -0
ultralytics/solutions/distance_calculation.py +124 -0
ultralytics/solutions/heatmap.py +127 -0
ultralytics/solutions/instance_segmentation.py +84 -0
ultralytics/solutions/object_blurrer.py +90 -0
ultralytics/solutions/object_counter.py +195 -0
ultralytics/solutions/object_cropper.py +84 -0
ultralytics/solutions/parking_management.py +273 -0
ultralytics/solutions/queue_management.py +93 -0
ultralytics/solutions/region_counter.py +120 -0
ultralytics/solutions/security_alarm.py +154 -0
ultralytics/solutions/similarity_search.py +172 -0
ultralytics/solutions/solutions.py +724 -0
ultralytics/solutions/speed_estimation.py +110 -0
ultralytics/solutions/streamlit_inference.py +196 -0
ultralytics/solutions/templates/similarity-search.html +160 -0
ultralytics/solutions/trackzone.py +88 -0
ultralytics/solutions/vision_eye.py +68 -0
ultralytics/trackers/__init__.py +7 -0
ultralytics/trackers/basetrack.py +124 -0
ultralytics/trackers/bot_sort.py +260 -0
ultralytics/trackers/byte_tracker.py +480 -0
ultralytics/trackers/track.py +125 -0
ultralytics/trackers/utils/__init__.py +1 -0
ultralytics/trackers/utils/gmc.py +376 -0
ultralytics/trackers/utils/kalman_filter.py +493 -0
ultralytics/trackers/utils/matching.py +157 -0
ultralytics/utils/__init__.py +1435 -0
ultralytics/utils/autobatch.py +106 -0
ultralytics/utils/autodevice.py +174 -0
ultralytics/utils/benchmarks.py +695 -0
ultralytics/utils/callbacks/__init__.py +5 -0
ultralytics/utils/callbacks/base.py +234 -0
ultralytics/utils/callbacks/clearml.py +153 -0
ultralytics/utils/callbacks/comet.py +552 -0
ultralytics/utils/callbacks/dvc.py +205 -0
ultralytics/utils/callbacks/hub.py +108 -0
ultralytics/utils/callbacks/mlflow.py +138 -0
ultralytics/utils/callbacks/neptune.py +140 -0
ultralytics/utils/callbacks/raytune.py +43 -0
ultralytics/utils/callbacks/tensorboard.py +132 -0
ultralytics/utils/callbacks/wb.py +185 -0
ultralytics/utils/checks.py +897 -0
ultralytics/utils/dist.py +119 -0
ultralytics/utils/downloads.py +499 -0
ultralytics/utils/errors.py +43 -0
ultralytics/utils/export.py +219 -0
ultralytics/utils/files.py +221 -0
ultralytics/utils/instance.py +499 -0
ultralytics/utils/loss.py +813 -0
ultralytics/utils/metrics.py +1356 -0
ultralytics/utils/ops.py +885 -0
ultralytics/utils/patches.py +143 -0
ultralytics/utils/plotting.py +1011 -0
ultralytics/utils/tal.py +416 -0
ultralytics/utils/torch_utils.py +990 -0
ultralytics/utils/triton.py +116 -0
ultralytics/utils/tuner.py +159 -0

ultralytics/models/yolo/segment/val.py ADDED Viewed

@@ -0,0 +1,428 @@
+# Ultralytics 🚀 AGPL-3.0 License - https://ultralytics.com/license
+from multiprocessing.pool import ThreadPool
+from pathlib import Path
+import numpy as np
+import torch
+import torch.nn.functional as F
+from ultralytics.models.yolo.detect import DetectionValidator
+from ultralytics.utils import LOGGER, NUM_THREADS, ops
+from ultralytics.utils.checks import check_requirements
+from ultralytics.utils.metrics import SegmentMetrics, box_iou, mask_iou
+from ultralytics.utils.plotting import output_to_target, plot_images
+class SegmentationValidator(DetectionValidator):
+    """
+    A class extending the DetectionValidator class for validation based on a segmentation model.
+    This validator handles the evaluation of segmentation models, processing both bounding box and mask predictions
+    to compute metrics such as mAP for both detection and segmentation tasks.
+    Attributes:
+        plot_masks (list): List to store masks for plotting.
+        process (callable): Function to process masks based on save_json and save_txt flags.
+        args (namespace): Arguments for the validator.
+        metrics (SegmentMetrics): Metrics calculator for segmentation tasks.
+        stats (dict): Dictionary to store statistics during validation.
+    Examples:
+        >>> from ultralytics.models.yolo.segment import SegmentationValidator
+        >>> args = dict(model="yolo11n-seg.pt", data="coco8-seg.yaml")
+        >>> validator = SegmentationValidator(args=args)
+        >>> validator()
+    """
+    def __init__(self, dataloader=None, save_dir=None, pbar=None, args=None, _callbacks=None):
+        """
+        Initialize SegmentationValidator and set task to 'segment', metrics to SegmentMetrics.
+        Args:
+            dataloader (torch.utils.data.DataLoader, optional): Dataloader to use for validation.
+            save_dir (Path, optional): Directory to save results.
+            pbar (Any, optional): Progress bar for displaying progress.
+            args (namespace, optional): Arguments for the validator.
+            _callbacks (list, optional): List of callback functions.
+        """
+        super().__init__(dataloader, save_dir, pbar, args, _callbacks)
+        self.plot_masks = None
+        self.process = None
+        self.args.task = "segment"
+        self.metrics = SegmentMetrics(save_dir=self.save_dir)
+    def preprocess(self, batch):
+        """Preprocess batch by converting masks to float and sending to device."""
+        batch = super().preprocess(batch)
+        batch["masks"] = batch["masks"].to(self.device).float()
+        return batch
+    def init_metrics(self, model):
+        """
+        Initialize metrics and select mask processing function based on save_json flag.
+        Args:
+            model (torch.nn.Module): Model to validate.
+        """
+        super().init_metrics(model)
+        self.plot_masks = []
+        if self.args.save_json:
+            check_requirements("pycocotools>=2.0.6")
+        # more accurate vs faster
+        self.process = ops.process_mask_native if self.args.save_json or self.args.save_txt else ops.process_mask
+        self.stats = dict(tp_m=[], tp=[], conf=[], pred_cls=[], target_cls=[], target_img=[])
+    def get_desc(self):
+        """Return a formatted description of evaluation metrics."""
+        return ("%22s" + "%11s" * 10) % (
+            "Class",
+            "Images",
+            "Instances",
+            "Box(P",
+            "R",
+            "mAP50",
+            "mAP50-95)",
+            "Mask(P",
+            "R",
+            "mAP50",
+            "mAP50-95)",
+        )
+    def postprocess(self, preds):
+        """
+        Post-process YOLO predictions and return output detections with proto.
+        Args:
+            preds (list): Raw predictions from the model.
+        Returns:
+            p (torch.Tensor): Processed detection predictions.
+            proto (torch.Tensor): Prototype masks for segmentation.
+        """
+        p = super().postprocess(preds[0])
+        proto = preds[1][-1] if len(preds[1]) == 3 else preds[1]  # second output is len 3 if pt, but only 1 if exported
+        return p, proto
+    def _prepare_batch(self, si, batch):
+        """
+        Prepare a batch for training or inference by processing images and targets.
+        Args:
+            si (int): Batch index.
+            batch (dict): Batch data containing images and targets.
+        Returns:
+            (dict): Prepared batch with processed images and targets.
+        """
+        prepared_batch = super()._prepare_batch(si, batch)
+        midx = [si] if self.args.overlap_mask else batch["batch_idx"] == si
+        prepared_batch["masks"] = batch["masks"][midx]
+        return prepared_batch
+    def _prepare_pred(self, pred, pbatch, proto):
+        """
+        Prepare predictions for evaluation by processing bounding boxes and masks.
+        Args:
+            pred (torch.Tensor): Raw predictions from the model.
+            pbatch (dict): Prepared batch data.
+            proto (torch.Tensor): Prototype masks for segmentation.
+        Returns:
+            predn (torch.Tensor): Processed bounding box predictions.
+            pred_masks (torch.Tensor): Processed mask predictions.
+        """
+        predn = super()._prepare_pred(pred, pbatch)
+        pred_masks = self.process(proto, pred[:, 6:], pred[:, :4], shape=pbatch["imgsz"])
+        return predn, pred_masks
+    def update_metrics(self, preds, batch):
+        """
+        Update metrics with the current batch predictions and targets.
+        Args:
+            preds (list): Predictions from the model.
+            batch (dict): Batch data containing images and targets.
+        """
+        for si, (pred, proto) in enumerate(zip(preds[0], preds[1])):
+            self.seen += 1
+            npr = len(pred)
+            stat = dict(
+                conf=torch.zeros(0, device=self.device),
+                pred_cls=torch.zeros(0, device=self.device),
+                tp=torch.zeros(npr, self.niou, dtype=torch.bool, device=self.device),
+                tp_m=torch.zeros(npr, self.niou, dtype=torch.bool, device=self.device),
+            )
+            pbatch = self._prepare_batch(si, batch)
+            cls, bbox = pbatch.pop("cls"), pbatch.pop("bbox")
+            nl = len(cls)
+            stat["target_cls"] = cls
+            stat["target_img"] = cls.unique()
+            if npr == 0:
+                if nl:
+                    for k in self.stats.keys():
+                        self.stats[k].append(stat[k])
+                    if self.args.plots:
+                        self.confusion_matrix.process_batch(detections=None, gt_bboxes=bbox, gt_cls=cls)
+                continue
+            # Masks
+            gt_masks = pbatch.pop("masks")
+            # Predictions
+            if self.args.single_cls:
+                pred[:, 5] = 0
+            predn, pred_masks = self._prepare_pred(pred, pbatch, proto)
+            stat["conf"] = predn[:, 4]
+            stat["pred_cls"] = predn[:, 5]
+            # Evaluate
+            if nl:
+                stat["tp"] = self._process_batch(predn, bbox, cls)
+                stat["tp_m"] = self._process_batch(
+                    predn, bbox, cls, pred_masks, gt_masks, self.args.overlap_mask, masks=True
+                )
+            if self.args.plots:
+                self.confusion_matrix.process_batch(predn, bbox, cls)
+            for k in self.stats.keys():
+                self.stats[k].append(stat[k])
+            pred_masks = torch.as_tensor(pred_masks, dtype=torch.uint8)
+            if self.args.plots and self.batch_i < 3:
+                self.plot_masks.append(pred_masks[:50].cpu())  # Limit plotted items for speed
+                if pred_masks.shape[0] > 50:
+                    LOGGER.warning("Limiting validation plots to first 50 items per image for speed...")
+            # Save
+            if self.args.save_json:
+                self.pred_to_json(
+                    predn,
+                    batch["im_file"][si],
+                    ops.scale_image(
+                        pred_masks.permute(1, 2, 0).contiguous().cpu().numpy(),
+                        pbatch["ori_shape"],
+                        ratio_pad=batch["ratio_pad"][si],
+                    ),
+                )
+            if self.args.save_txt:
+                self.save_one_txt(
+                    predn,
+                    pred_masks,
+                    self.args.save_conf,
+                    pbatch["ori_shape"],
+                    self.save_dir / "labels" / f"{Path(batch['im_file'][si]).stem}.txt",
+                )
+    def finalize_metrics(self, *args, **kwargs):
+        """
+        Finalize evaluation metrics by setting the speed attribute in the metrics object.
+        This method is called at the end of validation to set the processing speed for the metrics calculations.
+        It transfers the validator's speed measurement to the metrics object for reporting.
+        Args:
+            *args (Any): Variable length argument list.
+            **kwargs (Any): Arbitrary keyword arguments.
+        """
+        self.metrics.speed = self.speed
+        self.metrics.confusion_matrix = self.confusion_matrix
+    def _process_batch(self, detections, gt_bboxes, gt_cls, pred_masks=None, gt_masks=None, overlap=False, masks=False):
+        """
+        Compute correct prediction matrix for a batch based on bounding boxes and optional masks.
+        Args:
+            detections (torch.Tensor): Tensor of shape (N, 6) representing detected bounding boxes and
+                associated confidence scores and class indices. Each row is of the format [x1, y1, x2, y2, conf, class].
+            gt_bboxes (torch.Tensor): Tensor of shape (M, 4) representing ground truth bounding box coordinates.
+                Each row is of the format [x1, y1, x2, y2].
+            gt_cls (torch.Tensor): Tensor of shape (M,) representing ground truth class indices.
+            pred_masks (torch.Tensor, optional): Tensor representing predicted masks, if available. The shape should
+                match the ground truth masks.
+            gt_masks (torch.Tensor, optional): Tensor of shape (M, H, W) representing ground truth masks, if available.
+            overlap (bool): Flag indicating if overlapping masks should be considered.
+            masks (bool): Flag indicating if the batch contains mask data.
+        Returns:
+            (torch.Tensor): A correct prediction matrix of shape (N, 10), where 10 represents different IoU levels.
+        Note:
+            - If `masks` is True, the function computes IoU between predicted and ground truth masks.
+            - If `overlap` is True and `masks` is True, overlapping masks are taken into account when computing IoU.
+        Examples:
+            >>> detections = torch.tensor([[25, 30, 200, 300, 0.8, 1], [50, 60, 180, 290, 0.75, 0]])
+            >>> gt_bboxes = torch.tensor([[24, 29, 199, 299], [55, 65, 185, 295]])
+            >>> gt_cls = torch.tensor([1, 0])
+            >>> correct_preds = validator._process_batch(detections, gt_bboxes, gt_cls)
+        """
+        if masks:
+            if overlap:
+                nl = len(gt_cls)
+                index = torch.arange(nl, device=gt_masks.device).view(nl, 1, 1) + 1
+                gt_masks = gt_masks.repeat(nl, 1, 1)  # shape(1,640,640) -> (n,640,640)
+                gt_masks = torch.where(gt_masks == index, 1.0, 0.0)
+            if gt_masks.shape[1:] != pred_masks.shape[1:]:
+                gt_masks = F.interpolate(gt_masks[None], pred_masks.shape[1:], mode="bilinear", align_corners=False)[0]
+                gt_masks = gt_masks.gt_(0.5)
+            iou = mask_iou(gt_masks.view(gt_masks.shape[0], -1), pred_masks.view(pred_masks.shape[0], -1))
+        else:  # boxes
+            iou = box_iou(gt_bboxes, detections[:, :4])
+        return self.match_predictions(detections[:, 5], gt_cls, iou)
+    def plot_val_samples(self, batch, ni):
+        """
+        Plot validation samples with bounding box labels and masks.
+        Args:
+            batch (dict): Batch data containing images and targets.
+            ni (int): Batch index.
+        """
+        plot_images(
+            batch["img"],
+            batch["batch_idx"],
+            batch["cls"].squeeze(-1),
+            batch["bboxes"],
+            masks=batch["masks"],
+            paths=batch["im_file"],
+            fname=self.save_dir / f"val_batch{ni}_labels.jpg",
+            names=self.names,
+            on_plot=self.on_plot,
+        )
+    def plot_predictions(self, batch, preds, ni):
+        """
+        Plot batch predictions with masks and bounding boxes.
+        Args:
+            batch (dict): Batch data containing images.
+            preds (list): Predictions from the model.
+            ni (int): Batch index.
+        """
+        plot_images(
+            batch["img"],
+            *output_to_target(preds[0], max_det=50),  # not set to self.args.max_det due to slow plotting speed
+            torch.cat(self.plot_masks, dim=0) if len(self.plot_masks) else self.plot_masks,
+            paths=batch["im_file"],
+            fname=self.save_dir / f"val_batch{ni}_pred.jpg",
+            names=self.names,
+            on_plot=self.on_plot,
+        )  # pred
+        self.plot_masks.clear()
+    def save_one_txt(self, predn, pred_masks, save_conf, shape, file):
+        """
+        Save YOLO detections to a txt file in normalized coordinates in a specific format.
+        Args:
+            predn (torch.Tensor): Predictions in the format [x1, y1, x2, y2, conf, cls].
+            pred_masks (torch.Tensor): Predicted masks.
+            save_conf (bool): Whether to save confidence scores.
+            shape (tuple): Original image shape.
+            file (Path): File path to save the detections.
+        """
+        from ultralytics.engine.results import Results
+        Results(
+            np.zeros((shape[0], shape[1]), dtype=np.uint8),
+            path=None,
+            names=self.names,
+            boxes=predn[:, :6],
+            masks=pred_masks,
+        ).save_txt(file, save_conf=save_conf)
+    def pred_to_json(self, predn, filename, pred_masks):
+        """
+        Save one JSON result for COCO evaluation.
+        Args:
+            predn (torch.Tensor): Predictions in the format [x1, y1, x2, y2, conf, cls].
+            filename (str): Image filename.
+            pred_masks (numpy.ndarray): Predicted masks.
+        Examples:
+             >>> result = {"image_id": 42, "category_id": 18, "bbox": [258.15, 41.29, 348.26, 243.78], "score": 0.236}
+        """
+        from pycocotools.mask import encode  # noqa
+        def single_encode(x):
+            """Encode predicted masks as RLE and append results to jdict."""
+            rle = encode(np.asarray(x[:, :, None], order="F", dtype="uint8"))[0]
+            rle["counts"] = rle["counts"].decode("utf-8")
+            return rle
+        stem = Path(filename).stem
+        image_id = int(stem) if stem.isnumeric() else stem
+        box = ops.xyxy2xywh(predn[:, :4])  # xywh
+        box[:, :2] -= box[:, 2:] / 2  # xy center to top-left corner
+        pred_masks = np.transpose(pred_masks, (2, 0, 1))
+        with ThreadPool(NUM_THREADS) as pool:
+            rles = pool.map(single_encode, pred_masks)
+        for i, (p, b) in enumerate(zip(predn.tolist(), box.tolist())):
+            self.jdict.append(
+                {
+                    "image_id": image_id,
+                    "category_id": self.class_map[int(p[5])],
+                    "bbox": [round(x, 3) for x in b],
+                    "score": round(p[4], 5),
+                    "segmentation": rles[i],
+                }
+            )
+    def eval_json(self, stats):
+        """Return COCO-style object detection evaluation metrics."""
+        if self.args.save_json and (self.is_lvis or self.is_coco) and len(self.jdict):
+            pred_json = self.save_dir / "predictions.json"  # predictions
+            anno_json = (
+                self.data["path"]
+                / "annotations"
+                / ("instances_val2017.json" if self.is_coco else f"lvis_v1_{self.args.split}.json")
+            )  # annotations
+            pkg = "pycocotools" if self.is_coco else "lvis"
+            LOGGER.info(f"\nEvaluating {pkg} mAP using {pred_json} and {anno_json}...")
+            try:  # https://github.com/cocodataset/cocoapi/blob/master/PythonAPI/pycocoEvalDemo.ipynb
+                for x in anno_json, pred_json:
+                    assert x.is_file(), f"{x} file not found"
+                check_requirements("pycocotools>=2.0.6" if self.is_coco else "lvis>=0.5.3")
+                if self.is_coco:
+                    from pycocotools.coco import COCO  # noqa
+                    from pycocotools.cocoeval import COCOeval  # noqa
+                    anno = COCO(str(anno_json))  # init annotations api
+                    pred = anno.loadRes(str(pred_json))  # init predictions api (must pass string, not Path)
+                    vals = [COCOeval(anno, pred, "bbox"), COCOeval(anno, pred, "segm")]
+                else:
+                    from lvis import LVIS, LVISEval
+                    anno = LVIS(str(anno_json))
+                    pred = anno._load_json(str(pred_json))
+                    vals = [LVISEval(anno, pred, "bbox"), LVISEval(anno, pred, "segm")]
+                for i, eval in enumerate(vals):
+                    eval.params.imgIds = [int(Path(x).stem) for x in self.dataloader.dataset.im_files]  # im to eval
+                    eval.evaluate()
+                    eval.accumulate()
+                    eval.summarize()
+                    if self.is_lvis:
+                        eval.print_results()
+                    idx = i * 4 + 2
+                    # update mAP50-95 and mAP50
+                    stats[self.metrics.keys[idx + 1]], stats[self.metrics.keys[idx]] = (
+                        eval.stats[:2] if self.is_coco else [eval.results["AP"], eval.results["AP50"]]
+                    )
+                    if self.is_lvis:
+                        tag = "B" if i == 0 else "M"
+                        stats[f"metrics/APr({tag})"] = eval.results["APr"]
+                        stats[f"metrics/APc({tag})"] = eval.results["APc"]
+                        stats[f"metrics/APf({tag})"] = eval.results["APf"]
+                if self.is_lvis:
+                    stats["fitness"] = stats["metrics/mAP50-95(B)"]
+            except Exception as e:
+                LOGGER.warning(f"{pkg} unable to run: {e}")
+        return stats

ultralytics/models/yolo/world/__init__.py ADDED Viewed

@@ -0,0 +1,5 @@
+# Ultralytics 🚀 AGPL-3.0 License - https://ultralytics.com/license
+from .train import WorldTrainer
+__all__ = ["WorldTrainer"]

ultralytics/models/yolo/world/train.py ADDED Viewed

@@ -0,0 +1,119 @@
+# Ultralytics 🚀 AGPL-3.0 License - https://ultralytics.com/license
+import itertools
+from ultralytics.data import build_yolo_dataset
+from ultralytics.models import yolo
+from ultralytics.nn.tasks import WorldModel
+from ultralytics.utils import DEFAULT_CFG, RANK, checks
+from ultralytics.utils.torch_utils import de_parallel
+def on_pretrain_routine_end(trainer):
+    """Callback to set up model classes and text encoder at the end of the pretrain routine."""
+    if RANK in {-1, 0}:
+        # Set class names for evaluation
+        names = [name.split("/")[0] for name in list(trainer.test_loader.dataset.data["names"].values())]
+        de_parallel(trainer.ema.ema).set_classes(names, cache_clip_model=False)
+    device = next(trainer.model.parameters()).device
+    trainer.text_model, _ = trainer.clip.load("ViT-B/32", device=device)
+    for p in trainer.text_model.parameters():
+        p.requires_grad_(False)
+class WorldTrainer(yolo.detect.DetectionTrainer):
+    """
+    A class to fine-tune a world model on a close-set dataset.
+    This trainer extends the DetectionTrainer to support training YOLO World models, which combine
+    visual and textual features for improved object detection and understanding.
+    Attributes:
+        clip (module): The CLIP module for text-image understanding.
+        text_model (module): The text encoder model from CLIP.
+        model (WorldModel): The YOLO World model being trained.
+        data (dict): Dataset configuration containing class information.
+        args (dict): Training arguments and configuration.
+    Examples:
+        >>> from ultralytics.models.yolo.world import WorldModel
+        >>> args = dict(model="yolov8s-world.pt", data="coco8.yaml", epochs=3)
+        >>> trainer = WorldTrainer(overrides=args)
+        >>> trainer.train()
+    """
+    def __init__(self, cfg=DEFAULT_CFG, overrides=None, _callbacks=None):
+        """
+        Initialize a WorldTrainer object with given arguments.
+        Args:
+            cfg (dict): Configuration for the trainer.
+            overrides (dict, optional): Configuration overrides.
+            _callbacks (list, optional): List of callback functions.
+        """
+        if overrides is None:
+            overrides = {}
+        super().__init__(cfg, overrides, _callbacks)
+        # Import and assign clip
+        try:
+            import clip
+        except ImportError:
+            checks.check_requirements("git+https://github.com/ultralytics/CLIP.git")
+            import clip
+        self.clip = clip
+    def get_model(self, cfg=None, weights=None, verbose=True):
+        """
+        Return WorldModel initialized with specified config and weights.
+        Args:
+            cfg (Dict | str, optional): Model configuration.
+            weights (str, optional): Path to pretrained weights.
+            verbose (bool): Whether to display model info.
+        Returns:
+            (WorldModel): Initialized WorldModel.
+        """
+        # NOTE: This `nc` here is the max number of different text samples in one image, rather than the actual `nc`.
+        # NOTE: Following the official config, nc hard-coded to 80 for now.
+        model = WorldModel(
+            cfg["yaml_file"] if isinstance(cfg, dict) else cfg,
+            ch=self.data["channels"],
+            nc=min(self.data["nc"], 80),
+            verbose=verbose and RANK == -1,
+        )
+        if weights:
+            model.load(weights)
+        self.add_callback("on_pretrain_routine_end", on_pretrain_routine_end)
+        return model
+    def build_dataset(self, img_path, mode="train", batch=None):
+        """
+        Build YOLO Dataset for training or validation.
+        Args:
+            img_path (str): Path to the folder containing images.
+            mode (str): `train` mode or `val` mode, users are able to customize different augmentations for each mode.
+            batch (int, optional): Size of batches, this is for `rect`.
+        Returns:
+            (Dataset): YOLO dataset configured for training or validation.
+        """
+        gs = max(int(de_parallel(self.model).stride.max() if self.model else 0), 32)
+        return build_yolo_dataset(
+            self.args, img_path, batch, self.data, mode=mode, rect=mode == "val", stride=gs, multi_modal=mode == "train"
+        )
+    def preprocess_batch(self, batch):
+        """Preprocess a batch of images and text for YOLOWorld training."""
+        batch = super().preprocess_batch(batch)
+        # Add text features
+        texts = list(itertools.chain(*batch["texts"]))
+        text_token = self.clip.tokenize(texts).to(batch["img"].device)
+        txt_feats = self.text_model.encode_text(text_token).to(dtype=batch["img"].dtype)  # torch.float32
+        txt_feats = txt_feats / txt_feats.norm(p=2, dim=-1, keepdim=True)
+        batch["txt_feats"] = txt_feats.reshape(len(batch["texts"]), -1, txt_feats.shape[-1])
+        return batch