torch_em.data.datasets.medical.mmotu

The MMOTU dataset contains annotations for ovarian tumor segmentation in 2d ultrasound and contrast-enhanced ultrasound (CEUS) images.

The dataset contains 2D B-mode ultrasound images (OTU_2D) and CEUS images (OTU_CEUS) of ovarian tumors collected at Beijing Shijitan Hospital, Capital Medical University, with pixel-wise tumor masks and global tumor-type labels.

This mirror of the dataset is located at https://doi.org/10.6084/m9.figshare.25058690.v2 (CC BY 4.0). The original dataset and code are at https://github.com/cv516Buaa/MMOTU_DS2Net. This dataset is from the publication https://doi.org/10.1016/j.patcog.2025.112311. Please cite it if you use this dataset for your research.

  1"""The MMOTU dataset contains annotations for ovarian tumor segmentation in 2d ultrasound
  2and contrast-enhanced ultrasound (CEUS) images.
  3
  4The dataset contains 2D B-mode ultrasound images (`OTU_2D`) and CEUS images (`OTU_CEUS`)
  5of ovarian tumors collected at Beijing Shijitan Hospital, Capital Medical University, with
  6pixel-wise tumor masks and global tumor-type labels.
  7
  8This mirror of the dataset is located at https://doi.org/10.6084/m9.figshare.25058690.v2
  9(CC BY 4.0). The original dataset and code are at https://github.com/cv516Buaa/MMOTU_DS2Net.
 10This dataset is from the publication https://doi.org/10.1016/j.patcog.2025.112311.
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15from glob import glob
 16from typing import Union, Tuple, Optional, Literal, List
 17
 18from torch.utils.data import Dataset, DataLoader
 19
 20import torch_em
 21
 22from .. import util
 23
 24
 25URL = "https://ndownloader.figshare.com/files/44222642"
 26CHECKSUM = "5343647807cf34b507b66acd752dd94de265c6d1a7fde9cf8aa441a7283cde3e"
 27
 28
 29def get_mmotu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 30    """Download the MMOTU dataset.
 31
 32    Args:
 33        path: Filepath to a folder where the data is downloaded for further processing.
 34        download: Whether to download the data if it is not present.
 35
 36    Returns:
 37        Filepath where the data is downloaded.
 38    """
 39    data_dir = os.path.join(path, "dataset")
 40    if os.path.exists(data_dir):
 41        return data_dir
 42
 43    os.makedirs(path, exist_ok=True)
 44
 45    zip_path = os.path.join(path, "dataset.zip")
 46    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 47    util.unzip(zip_path=zip_path, dst=path)
 48
 49    return data_dir
 50
 51
 52def get_mmotu_paths(
 53    path: Union[os.PathLike, str],
 54    modality: Optional[Literal["2d", "ceus"]] = None,
 55    split: Optional[Literal["train", "test"]] = None,
 56    download: bool = False,
 57) -> Tuple[List[str], List[str]]:
 58    """Get paths to the MMOTU data.
 59
 60    Args:
 61        path: Filepath to a folder where the data is downloaded for further processing.
 62        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
 63            or contrast-enhanced ultrasound ('ceus').
 64        split: The choice of data split, only valid for the 2d modality.
 65        download: Whether to download the data if it is not present.
 66
 67    Returns:
 68        List of filepaths for the image data.
 69        List of filepaths for the label data.
 70    """
 71    data_dir = get_mmotu_data(path=path, download=download)
 72
 73    if modality is None:
 74        modality = "*"
 75    elif modality not in ["2d", "ceus"]:
 76        raise ValueError(f"'{modality}' is not a valid modality choice.")
 77
 78    image_paths, gt_paths = [], []
 79
 80    if modality in ("2d", "*"):
 81        if split is None:
 82            split = "*"
 83        elif split not in ["train", "test"]:
 84            raise ValueError(f"'{split}' is not a valid split choice.")
 85
 86        if split in ("train", "*"):
 87            image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "train", "train_image", "*.JPG"))))
 88            gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "train", "train_label", "label", "*.PNG"))))
 89
 90        if split in ("test", "*"):
 91            image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "test", "image", "*.JPG"))))
 92            gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "test", "label", "black_write", "*.PNG"))))
 93
 94    if modality in ("ceus", "*"):
 95        image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_CEUS", "image", "*.JPG"))))
 96        gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_CEUS", "label", "*.PNG"))))
 97
 98    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
 99        raise RuntimeError("Something went wrong with fetching the image and label paths.")
100
101    return image_paths, gt_paths
102
103
104def get_mmotu_dataset(
105    path: Union[os.PathLike, str],
106    patch_shape: Tuple[int, int],
107    modality: Optional[Literal["2d", "ceus"]] = None,
108    split: Optional[Literal["train", "test"]] = None,
109    resize_inputs: bool = False,
110    download: bool = False,
111    **kwargs
112) -> Dataset:
113    """Get the MMOTU dataset for ovarian tumor segmentation.
114
115    Args:
116        path: Filepath to a folder where the data is downloaded for further processing.
117        patch_shape: The patch shape to use for training.
118        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
119            or contrast-enhanced ultrasound ('ceus').
120        split: The choice of data split, only valid for the 2d modality.
121        resize_inputs: Whether to resize the inputs.
122        download: Whether to download the data if it is not present.
123        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
124
125    Returns:
126        The segmentation dataset.
127    """
128    image_paths, gt_paths = get_mmotu_paths(path, modality, split, download)
129
130    if resize_inputs:
131        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
132        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
133            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
134        )
135
136    return torch_em.default_segmentation_dataset(
137        raw_paths=image_paths,
138        raw_key=None,
139        label_paths=gt_paths,
140        label_key=None,
141        patch_shape=patch_shape,
142        is_seg_dataset=False,
143        **kwargs
144    )
145
146
147def get_mmotu_loader(
148    path: Union[os.PathLike, str],
149    batch_size: int,
150    patch_shape: Tuple[int, int],
151    modality: Optional[Literal["2d", "ceus"]] = None,
152    split: Optional[Literal["train", "test"]] = None,
153    resize_inputs: bool = False,
154    download: bool = False,
155    **kwargs
156) -> DataLoader:
157    """Get the MMOTU dataloader for ovarian tumor segmentation.
158
159    Args:
160        path: Filepath to a folder where the data is downloaded for further processing.
161        batch_size: The batch size for training.
162        patch_shape: The patch shape to use for training.
163        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
164            or contrast-enhanced ultrasound ('ceus').
165        split: The choice of data split, only valid for the 2d modality.
166        resize_inputs: Whether to resize the inputs.
167        download: Whether to download the data if it is not present.
168        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
169
170    Returns:
171        The DataLoader.
172    """
173    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
174    dataset = get_mmotu_dataset(path, patch_shape, modality, split, resize_inputs, download, **ds_kwargs)
175    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://ndownloader.figshare.com/files/44222642'
CHECKSUM = '5343647807cf34b507b66acd752dd94de265c6d1a7fde9cf8aa441a7283cde3e'
def get_mmotu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
30def get_mmotu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31    """Download the MMOTU dataset.
32
33    Args:
34        path: Filepath to a folder where the data is downloaded for further processing.
35        download: Whether to download the data if it is not present.
36
37    Returns:
38        Filepath where the data is downloaded.
39    """
40    data_dir = os.path.join(path, "dataset")
41    if os.path.exists(data_dir):
42        return data_dir
43
44    os.makedirs(path, exist_ok=True)
45
46    zip_path = os.path.join(path, "dataset.zip")
47    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
48    util.unzip(zip_path=zip_path, dst=path)
49
50    return data_dir

Download the MMOTU dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_mmotu_paths( path: Union[os.PathLike, str], modality: Optional[Literal['2d', 'ceus']] = None, split: Optional[Literal['train', 'test']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 53def get_mmotu_paths(
 54    path: Union[os.PathLike, str],
 55    modality: Optional[Literal["2d", "ceus"]] = None,
 56    split: Optional[Literal["train", "test"]] = None,
 57    download: bool = False,
 58) -> Tuple[List[str], List[str]]:
 59    """Get paths to the MMOTU data.
 60
 61    Args:
 62        path: Filepath to a folder where the data is downloaded for further processing.
 63        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
 64            or contrast-enhanced ultrasound ('ceus').
 65        split: The choice of data split, only valid for the 2d modality.
 66        download: Whether to download the data if it is not present.
 67
 68    Returns:
 69        List of filepaths for the image data.
 70        List of filepaths for the label data.
 71    """
 72    data_dir = get_mmotu_data(path=path, download=download)
 73
 74    if modality is None:
 75        modality = "*"
 76    elif modality not in ["2d", "ceus"]:
 77        raise ValueError(f"'{modality}' is not a valid modality choice.")
 78
 79    image_paths, gt_paths = [], []
 80
 81    if modality in ("2d", "*"):
 82        if split is None:
 83            split = "*"
 84        elif split not in ["train", "test"]:
 85            raise ValueError(f"'{split}' is not a valid split choice.")
 86
 87        if split in ("train", "*"):
 88            image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "train", "train_image", "*.JPG"))))
 89            gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "train", "train_label", "label", "*.PNG"))))
 90
 91        if split in ("test", "*"):
 92            image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "test", "image", "*.JPG"))))
 93            gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_2D", "test", "label", "black_write", "*.PNG"))))
 94
 95    if modality in ("ceus", "*"):
 96        image_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_CEUS", "image", "*.JPG"))))
 97        gt_paths.extend(sorted(glob(os.path.join(data_dir, "OTU_CEUS", "label", "*.PNG"))))
 98
 99    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
100        raise RuntimeError("Something went wrong with fetching the image and label paths.")
101
102    return image_paths, gt_paths

Get paths to the MMOTU data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of imaging modality, either conventional 2d ultrasound ('2d') or contrast-enhanced ultrasound ('ceus').
  • split: The choice of data split, only valid for the 2d modality.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_mmotu_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], modality: Optional[Literal['2d', 'ceus']] = None, split: Optional[Literal['train', 'test']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
105def get_mmotu_dataset(
106    path: Union[os.PathLike, str],
107    patch_shape: Tuple[int, int],
108    modality: Optional[Literal["2d", "ceus"]] = None,
109    split: Optional[Literal["train", "test"]] = None,
110    resize_inputs: bool = False,
111    download: bool = False,
112    **kwargs
113) -> Dataset:
114    """Get the MMOTU dataset for ovarian tumor segmentation.
115
116    Args:
117        path: Filepath to a folder where the data is downloaded for further processing.
118        patch_shape: The patch shape to use for training.
119        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
120            or contrast-enhanced ultrasound ('ceus').
121        split: The choice of data split, only valid for the 2d modality.
122        resize_inputs: Whether to resize the inputs.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    image_paths, gt_paths = get_mmotu_paths(path, modality, split, download)
130
131    if resize_inputs:
132        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
133        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
134            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
135        )
136
137    return torch_em.default_segmentation_dataset(
138        raw_paths=image_paths,
139        raw_key=None,
140        label_paths=gt_paths,
141        label_key=None,
142        patch_shape=patch_shape,
143        is_seg_dataset=False,
144        **kwargs
145    )

Get the MMOTU dataset for ovarian tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of imaging modality, either conventional 2d ultrasound ('2d') or contrast-enhanced ultrasound ('ceus').
  • split: The choice of data split, only valid for the 2d modality.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_mmotu_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], modality: Optional[Literal['2d', 'ceus']] = None, split: Optional[Literal['train', 'test']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
148def get_mmotu_loader(
149    path: Union[os.PathLike, str],
150    batch_size: int,
151    patch_shape: Tuple[int, int],
152    modality: Optional[Literal["2d", "ceus"]] = None,
153    split: Optional[Literal["train", "test"]] = None,
154    resize_inputs: bool = False,
155    download: bool = False,
156    **kwargs
157) -> DataLoader:
158    """Get the MMOTU dataloader for ovarian tumor segmentation.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        batch_size: The batch size for training.
163        patch_shape: The patch shape to use for training.
164        modality: The choice of imaging modality, either conventional 2d ultrasound ('2d')
165            or contrast-enhanced ultrasound ('ceus').
166        split: The choice of data split, only valid for the 2d modality.
167        resize_inputs: Whether to resize the inputs.
168        download: Whether to download the data if it is not present.
169        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
170
171    Returns:
172        The DataLoader.
173    """
174    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
175    dataset = get_mmotu_dataset(path, patch_shape, modality, split, resize_inputs, download, **ds_kwargs)
176    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the MMOTU dataloader for ovarian tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of imaging modality, either conventional 2d ultrasound ('2d') or contrast-enhanced ultrasound ('ceus').
  • split: The choice of data split, only valid for the 2d modality.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.