torch_em.data.datasets.medical.cesarean_scar_defect

The Cesarean Scar Defect (CSD) dataset contains annotations for cesarean scar defect segmentation in transvaginal ultrasound images.

The dataset consists of 501 images with binary segmentation masks (foreground: scar defect), stored in a Pascal VOC-style layout ('JPEGImages' for the raw images, 'SegmentationClass' for the label masks). The official 'ImageSets/Segmentation/train.txt' and 'val.txt' files list more image ids than are actually shipped in the archive (802 and 507 respectively, out of 501 total images); this module intersects the listed ids with the images that are actually present on disk to build the 'train' (401 images) and 'val' (100 images) splits.

The data is located at https://doi.org/10.5281/zenodo.17789273, released under a CC-BY-4.0 license.

This dataset is from the publication https://arxiv.org/abs/2605.26774. Please cite it if you use this dataset for your research.

  1"""The Cesarean Scar Defect (CSD) dataset contains annotations for cesarean scar defect
  2segmentation in transvaginal ultrasound images.
  3
  4The dataset consists of 501 images with binary segmentation masks (foreground: scar defect),
  5stored in a Pascal VOC-style layout ('JPEGImages' for the raw images, 'SegmentationClass' for
  6the label masks). The official 'ImageSets/Segmentation/train.txt' and 'val.txt' files list more
  7image ids than are actually shipped in the archive (802 and 507 respectively, out of 501 total
  8images); this module intersects the listed ids with the images that are actually present on disk
  9to build the 'train' (401 images) and 'val' (100 images) splits.
 10
 11The data is located at https://doi.org/10.5281/zenodo.17789273, released under a CC-BY-4.0 license.
 12
 13This dataset is from the publication https://arxiv.org/abs/2605.26774.
 14Please cite it if you use this dataset for your research.
 15"""
 16
 17import os
 18from glob import glob
 19from natsort import natsorted
 20from typing import Union, Tuple, Literal, List
 21
 22from torch.utils.data import Dataset, DataLoader
 23
 24import torch_em
 25
 26from .. import util
 27
 28
 29URL = "https://zenodo.org/records/17789273/files/VOCdevkit_CSD.zip"
 30CHECKSUM = "d3c37dd971260d7be2e50e07375f199a2da6e103141f4e4dc06789f25f983380"
 31
 32SPLITS = ("train", "val")
 33
 34
 35def get_cesarean_scar_defect_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 36    """Download the Cesarean Scar Defect dataset.
 37
 38    Args:
 39        path: Filepath to a folder where the data is downloaded for further processing.
 40        download: Whether to download the data if it is not present.
 41
 42    Returns:
 43        Filepath to the VOC-style data folder.
 44    """
 45    data_dir = os.path.join(path, "VOCdevkit_CSD", "VOCdevkit_CSD", "VOC2007")
 46    if os.path.exists(data_dir):
 47        return data_dir
 48
 49    os.makedirs(path, exist_ok=True)
 50
 51    zip_path = os.path.join(path, "VOCdevkit_CSD.zip")
 52    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 53    util.unzip(zip_path=zip_path, dst=path)
 54
 55    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder '{data_dir}'."
 56
 57    return data_dir
 58
 59
 60def get_cesarean_scar_defect_paths(
 61    path: Union[os.PathLike, str], split: Literal["train", "val"], download: bool = False,
 62) -> Tuple[List[str], List[str]]:
 63    """Get paths to the Cesarean Scar Defect data.
 64
 65    Args:
 66        path: Filepath to a folder where the data is downloaded for further processing.
 67        split: The choice of data split. Either 'train' or 'val'.
 68        download: Whether to download the data if it is not present.
 69
 70    Returns:
 71        List of filepaths for the image data.
 72        List of filepaths for the label data.
 73    """
 74    if split not in SPLITS:
 75        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
 76
 77    data_dir = get_cesarean_scar_defect_data(path, download)
 78
 79    with open(os.path.join(data_dir, "ImageSets", "Segmentation", f"{split}.txt")) as f:
 80        split_ids = {line.strip() for line in f if line.strip()}
 81
 82    label_paths = natsorted(
 83        p for p in glob(os.path.join(data_dir, "SegmentationClass", "*.png"))
 84        if os.path.splitext(os.path.basename(p))[0] in split_ids
 85    )
 86    raw_paths = [
 87        os.path.join(data_dir, "JPEGImages", f"{os.path.splitext(os.path.basename(p))[0]}.jpg") for p in label_paths
 88    ]
 89
 90    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 91    assert all(os.path.exists(p) for p in raw_paths)
 92
 93    return raw_paths, label_paths
 94
 95
 96def get_cesarean_scar_defect_dataset(
 97    path: Union[os.PathLike, str],
 98    patch_shape: Tuple[int, int],
 99    split: Literal["train", "val"],
100    resize_inputs: bool = False,
101    download: bool = False,
102    **kwargs
103) -> Dataset:
104    """Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound.
105
106    Args:
107        path: Filepath to a folder where the data is downloaded for further processing.
108        patch_shape: The patch shape to use for training.
109        split: The choice of data split. Either 'train' or 'val'.
110        resize_inputs: Whether to resize the inputs to the patch shape.
111        download: Whether to download the data if it is not present.
112        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
113
114    Returns:
115        The segmentation dataset.
116    """
117    raw_paths, label_paths = get_cesarean_scar_defect_paths(path, split, download)
118
119    if resize_inputs:
120        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
121        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
122            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
123        )
124
125    return torch_em.default_segmentation_dataset(
126        raw_paths=raw_paths,
127        raw_key=None,
128        label_paths=label_paths,
129        label_key=None,
130        is_seg_dataset=False,
131        patch_shape=patch_shape,
132        **kwargs
133    )
134
135
136def get_cesarean_scar_defect_loader(
137    path: Union[os.PathLike, str],
138    batch_size: int,
139    patch_shape: Tuple[int, int],
140    split: Literal["train", "val"],
141    resize_inputs: bool = False,
142    download: bool = False,
143    **kwargs
144) -> DataLoader:
145    """Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound.
146
147    Args:
148        path: Filepath to a folder where the data is downloaded for further processing.
149        batch_size: The batch size for training.
150        patch_shape: The patch shape to use for training.
151        split: The choice of data split. Either 'train' or 'val'.
152        resize_inputs: Whether to resize the inputs to the patch shape.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
155
156    Returns:
157        The DataLoader.
158    """
159    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
160    dataset = get_cesarean_scar_defect_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
161    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/17789273/files/VOCdevkit_CSD.zip'
CHECKSUM = 'd3c37dd971260d7be2e50e07375f199a2da6e103141f4e4dc06789f25f983380'
SPLITS = ('train', 'val')
def get_cesarean_scar_defect_data(path: Union[os.PathLike, str], download: bool = False) -> str:
36def get_cesarean_scar_defect_data(path: Union[os.PathLike, str], download: bool = False) -> str:
37    """Download the Cesarean Scar Defect dataset.
38
39    Args:
40        path: Filepath to a folder where the data is downloaded for further processing.
41        download: Whether to download the data if it is not present.
42
43    Returns:
44        Filepath to the VOC-style data folder.
45    """
46    data_dir = os.path.join(path, "VOCdevkit_CSD", "VOCdevkit_CSD", "VOC2007")
47    if os.path.exists(data_dir):
48        return data_dir
49
50    os.makedirs(path, exist_ok=True)
51
52    zip_path = os.path.join(path, "VOCdevkit_CSD.zip")
53    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
54    util.unzip(zip_path=zip_path, dst=path)
55
56    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder '{data_dir}'."
57
58    return data_dir

Download the Cesarean Scar Defect dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the VOC-style data folder.

def get_cesarean_scar_defect_paths( path: Union[os.PathLike, str], split: Literal['train', 'val'], download: bool = False) -> Tuple[List[str], List[str]]:
61def get_cesarean_scar_defect_paths(
62    path: Union[os.PathLike, str], split: Literal["train", "val"], download: bool = False,
63) -> Tuple[List[str], List[str]]:
64    """Get paths to the Cesarean Scar Defect data.
65
66    Args:
67        path: Filepath to a folder where the data is downloaded for further processing.
68        split: The choice of data split. Either 'train' or 'val'.
69        download: Whether to download the data if it is not present.
70
71    Returns:
72        List of filepaths for the image data.
73        List of filepaths for the label data.
74    """
75    if split not in SPLITS:
76        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
77
78    data_dir = get_cesarean_scar_defect_data(path, download)
79
80    with open(os.path.join(data_dir, "ImageSets", "Segmentation", f"{split}.txt")) as f:
81        split_ids = {line.strip() for line in f if line.strip()}
82
83    label_paths = natsorted(
84        p for p in glob(os.path.join(data_dir, "SegmentationClass", "*.png"))
85        if os.path.splitext(os.path.basename(p))[0] in split_ids
86    )
87    raw_paths = [
88        os.path.join(data_dir, "JPEGImages", f"{os.path.splitext(os.path.basename(p))[0]}.jpg") for p in label_paths
89    ]
90
91    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
92    assert all(os.path.exists(p) for p in raw_paths)
93
94    return raw_paths, label_paths

Get paths to the Cesarean Scar Defect data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. Either 'train' or 'val'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_cesarean_scar_defect_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 97def get_cesarean_scar_defect_dataset(
 98    path: Union[os.PathLike, str],
 99    patch_shape: Tuple[int, int],
100    split: Literal["train", "val"],
101    resize_inputs: bool = False,
102    download: bool = False,
103    **kwargs
104) -> Dataset:
105    """Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound.
106
107    Args:
108        path: Filepath to a folder where the data is downloaded for further processing.
109        patch_shape: The patch shape to use for training.
110        split: The choice of data split. Either 'train' or 'val'.
111        resize_inputs: Whether to resize the inputs to the patch shape.
112        download: Whether to download the data if it is not present.
113        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
114
115    Returns:
116        The segmentation dataset.
117    """
118    raw_paths, label_paths = get_cesarean_scar_defect_paths(path, split, download)
119
120    if resize_inputs:
121        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
122        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
123            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
124        )
125
126    return torch_em.default_segmentation_dataset(
127        raw_paths=raw_paths,
128        raw_key=None,
129        label_paths=label_paths,
130        label_key=None,
131        is_seg_dataset=False,
132        patch_shape=patch_shape,
133        **kwargs
134    )

Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train' or 'val'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_cesarean_scar_defect_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
137def get_cesarean_scar_defect_loader(
138    path: Union[os.PathLike, str],
139    batch_size: int,
140    patch_shape: Tuple[int, int],
141    split: Literal["train", "val"],
142    resize_inputs: bool = False,
143    download: bool = False,
144    **kwargs
145) -> DataLoader:
146    """Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound.
147
148    Args:
149        path: Filepath to a folder where the data is downloaded for further processing.
150        batch_size: The batch size for training.
151        patch_shape: The patch shape to use for training.
152        split: The choice of data split. Either 'train' or 'val'.
153        resize_inputs: Whether to resize the inputs to the patch shape.
154        download: Whether to download the data if it is not present.
155        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
156
157    Returns:
158        The DataLoader.
159    """
160    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
161    dataset = get_cesarean_scar_defect_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
162    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train' or 'val'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.