torch_em.data.datasets.medical.cesarean_scar_defect
The Cesarean Scar Defect (CSD) dataset contains annotations for cesarean scar defect segmentation in transvaginal ultrasound images.
The dataset consists of 501 images with binary segmentation masks (foreground: scar defect), stored in a Pascal VOC-style layout ('JPEGImages' for the raw images, 'SegmentationClass' for the label masks). The official 'ImageSets/Segmentation/train.txt' and 'val.txt' files list more image ids than are actually shipped in the archive (802 and 507 respectively, out of 501 total images); this module intersects the listed ids with the images that are actually present on disk to build the 'train' (401 images) and 'val' (100 images) splits.
The data is located at https://doi.org/10.5281/zenodo.17789273, released under a CC-BY-4.0 license.
This dataset is from the publication https://arxiv.org/abs/2605.26774. Please cite it if you use this dataset for your research.
1"""The Cesarean Scar Defect (CSD) dataset contains annotations for cesarean scar defect 2segmentation in transvaginal ultrasound images. 3 4The dataset consists of 501 images with binary segmentation masks (foreground: scar defect), 5stored in a Pascal VOC-style layout ('JPEGImages' for the raw images, 'SegmentationClass' for 6the label masks). The official 'ImageSets/Segmentation/train.txt' and 'val.txt' files list more 7image ids than are actually shipped in the archive (802 and 507 respectively, out of 501 total 8images); this module intersects the listed ids with the images that are actually present on disk 9to build the 'train' (401 images) and 'val' (100 images) splits. 10 11The data is located at https://doi.org/10.5281/zenodo.17789273, released under a CC-BY-4.0 license. 12 13This dataset is from the publication https://arxiv.org/abs/2605.26774. 14Please cite it if you use this dataset for your research. 15""" 16 17import os 18from glob import glob 19from natsort import natsorted 20from typing import Union, Tuple, Literal, List 21 22from torch.utils.data import Dataset, DataLoader 23 24import torch_em 25 26from .. import util 27 28 29URL = "https://zenodo.org/records/17789273/files/VOCdevkit_CSD.zip" 30CHECKSUM = "d3c37dd971260d7be2e50e07375f199a2da6e103141f4e4dc06789f25f983380" 31 32SPLITS = ("train", "val") 33 34 35def get_cesarean_scar_defect_data(path: Union[os.PathLike, str], download: bool = False) -> str: 36 """Download the Cesarean Scar Defect dataset. 37 38 Args: 39 path: Filepath to a folder where the data is downloaded for further processing. 40 download: Whether to download the data if it is not present. 41 42 Returns: 43 Filepath to the VOC-style data folder. 44 """ 45 data_dir = os.path.join(path, "VOCdevkit_CSD", "VOCdevkit_CSD", "VOC2007") 46 if os.path.exists(data_dir): 47 return data_dir 48 49 os.makedirs(path, exist_ok=True) 50 51 zip_path = os.path.join(path, "VOCdevkit_CSD.zip") 52 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 53 util.unzip(zip_path=zip_path, dst=path) 54 55 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder '{data_dir}'." 56 57 return data_dir 58 59 60def get_cesarean_scar_defect_paths( 61 path: Union[os.PathLike, str], split: Literal["train", "val"], download: bool = False, 62) -> Tuple[List[str], List[str]]: 63 """Get paths to the Cesarean Scar Defect data. 64 65 Args: 66 path: Filepath to a folder where the data is downloaded for further processing. 67 split: The choice of data split. Either 'train' or 'val'. 68 download: Whether to download the data if it is not present. 69 70 Returns: 71 List of filepaths for the image data. 72 List of filepaths for the label data. 73 """ 74 if split not in SPLITS: 75 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 76 77 data_dir = get_cesarean_scar_defect_data(path, download) 78 79 with open(os.path.join(data_dir, "ImageSets", "Segmentation", f"{split}.txt")) as f: 80 split_ids = {line.strip() for line in f if line.strip()} 81 82 label_paths = natsorted( 83 p for p in glob(os.path.join(data_dir, "SegmentationClass", "*.png")) 84 if os.path.splitext(os.path.basename(p))[0] in split_ids 85 ) 86 raw_paths = [ 87 os.path.join(data_dir, "JPEGImages", f"{os.path.splitext(os.path.basename(p))[0]}.jpg") for p in label_paths 88 ] 89 90 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 91 assert all(os.path.exists(p) for p in raw_paths) 92 93 return raw_paths, label_paths 94 95 96def get_cesarean_scar_defect_dataset( 97 path: Union[os.PathLike, str], 98 patch_shape: Tuple[int, int], 99 split: Literal["train", "val"], 100 resize_inputs: bool = False, 101 download: bool = False, 102 **kwargs 103) -> Dataset: 104 """Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound. 105 106 Args: 107 path: Filepath to a folder where the data is downloaded for further processing. 108 patch_shape: The patch shape to use for training. 109 split: The choice of data split. Either 'train' or 'val'. 110 resize_inputs: Whether to resize the inputs to the patch shape. 111 download: Whether to download the data if it is not present. 112 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 113 114 Returns: 115 The segmentation dataset. 116 """ 117 raw_paths, label_paths = get_cesarean_scar_defect_paths(path, split, download) 118 119 if resize_inputs: 120 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 121 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 122 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 123 ) 124 125 return torch_em.default_segmentation_dataset( 126 raw_paths=raw_paths, 127 raw_key=None, 128 label_paths=label_paths, 129 label_key=None, 130 is_seg_dataset=False, 131 patch_shape=patch_shape, 132 **kwargs 133 ) 134 135 136def get_cesarean_scar_defect_loader( 137 path: Union[os.PathLike, str], 138 batch_size: int, 139 patch_shape: Tuple[int, int], 140 split: Literal["train", "val"], 141 resize_inputs: bool = False, 142 download: bool = False, 143 **kwargs 144) -> DataLoader: 145 """Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound. 146 147 Args: 148 path: Filepath to a folder where the data is downloaded for further processing. 149 batch_size: The batch size for training. 150 patch_shape: The patch shape to use for training. 151 split: The choice of data split. Either 'train' or 'val'. 152 resize_inputs: Whether to resize the inputs to the patch shape. 153 download: Whether to download the data if it is not present. 154 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 155 156 Returns: 157 The DataLoader. 158 """ 159 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 160 dataset = get_cesarean_scar_defect_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 161 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
36def get_cesarean_scar_defect_data(path: Union[os.PathLike, str], download: bool = False) -> str: 37 """Download the Cesarean Scar Defect dataset. 38 39 Args: 40 path: Filepath to a folder where the data is downloaded for further processing. 41 download: Whether to download the data if it is not present. 42 43 Returns: 44 Filepath to the VOC-style data folder. 45 """ 46 data_dir = os.path.join(path, "VOCdevkit_CSD", "VOCdevkit_CSD", "VOC2007") 47 if os.path.exists(data_dir): 48 return data_dir 49 50 os.makedirs(path, exist_ok=True) 51 52 zip_path = os.path.join(path, "VOCdevkit_CSD.zip") 53 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 54 util.unzip(zip_path=zip_path, dst=path) 55 56 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder '{data_dir}'." 57 58 return data_dir
Download the Cesarean Scar Defect dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the VOC-style data folder.
61def get_cesarean_scar_defect_paths( 62 path: Union[os.PathLike, str], split: Literal["train", "val"], download: bool = False, 63) -> Tuple[List[str], List[str]]: 64 """Get paths to the Cesarean Scar Defect data. 65 66 Args: 67 path: Filepath to a folder where the data is downloaded for further processing. 68 split: The choice of data split. Either 'train' or 'val'. 69 download: Whether to download the data if it is not present. 70 71 Returns: 72 List of filepaths for the image data. 73 List of filepaths for the label data. 74 """ 75 if split not in SPLITS: 76 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 77 78 data_dir = get_cesarean_scar_defect_data(path, download) 79 80 with open(os.path.join(data_dir, "ImageSets", "Segmentation", f"{split}.txt")) as f: 81 split_ids = {line.strip() for line in f if line.strip()} 82 83 label_paths = natsorted( 84 p for p in glob(os.path.join(data_dir, "SegmentationClass", "*.png")) 85 if os.path.splitext(os.path.basename(p))[0] in split_ids 86 ) 87 raw_paths = [ 88 os.path.join(data_dir, "JPEGImages", f"{os.path.splitext(os.path.basename(p))[0]}.jpg") for p in label_paths 89 ] 90 91 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 92 assert all(os.path.exists(p) for p in raw_paths) 93 94 return raw_paths, label_paths
Get paths to the Cesarean Scar Defect data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train' or 'val'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
97def get_cesarean_scar_defect_dataset( 98 path: Union[os.PathLike, str], 99 patch_shape: Tuple[int, int], 100 split: Literal["train", "val"], 101 resize_inputs: bool = False, 102 download: bool = False, 103 **kwargs 104) -> Dataset: 105 """Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound. 106 107 Args: 108 path: Filepath to a folder where the data is downloaded for further processing. 109 patch_shape: The patch shape to use for training. 110 split: The choice of data split. Either 'train' or 'val'. 111 resize_inputs: Whether to resize the inputs to the patch shape. 112 download: Whether to download the data if it is not present. 113 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 114 115 Returns: 116 The segmentation dataset. 117 """ 118 raw_paths, label_paths = get_cesarean_scar_defect_paths(path, split, download) 119 120 if resize_inputs: 121 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 122 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 123 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 124 ) 125 126 return torch_em.default_segmentation_dataset( 127 raw_paths=raw_paths, 128 raw_key=None, 129 label_paths=label_paths, 130 label_key=None, 131 is_seg_dataset=False, 132 patch_shape=patch_shape, 133 **kwargs 134 )
Get the Cesarean Scar Defect dataset for scar defect segmentation in transvaginal ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'val'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
137def get_cesarean_scar_defect_loader( 138 path: Union[os.PathLike, str], 139 batch_size: int, 140 patch_shape: Tuple[int, int], 141 split: Literal["train", "val"], 142 resize_inputs: bool = False, 143 download: bool = False, 144 **kwargs 145) -> DataLoader: 146 """Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound. 147 148 Args: 149 path: Filepath to a folder where the data is downloaded for further processing. 150 batch_size: The batch size for training. 151 patch_shape: The patch shape to use for training. 152 split: The choice of data split. Either 'train' or 'val'. 153 resize_inputs: Whether to resize the inputs to the patch shape. 154 download: Whether to download the data if it is not present. 155 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 156 157 Returns: 158 The DataLoader. 159 """ 160 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 161 dataset = get_cesarean_scar_defect_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 162 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Cesarean Scar Defect dataloader for scar defect segmentation in transvaginal ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'val'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.