torch_em.data.datasets.medical.fugc
The FUGC dataset contains annotations for cervical lip segmentation in transvaginal ultrasound images.
The dataset consists of 890 transvaginal ultrasound images (336x544 pixels, RGB) from the Fetal Ultrasound Grand Challenge on semi-supervised cervical segmentation. The anterior and posterior cervical lips were segmented automatically and then manually corrected by an experienced radiologist.
Labeled images are available for three splits: 'train' (50), 'val' (90) and 'test' (300). The training split additionally contains 450 unlabeled images, which are not used by this loader. The label ids are 0 (background), 1 (anterior lip) and 2 (posterior lip). The record only names the two lips without listing the ids, so this order is inferred: the structure with id 1 lies above the one with id 2 in all 140 labeled training and validation images.
The data is located at https://doi.org/10.5281/zenodo.16893174, released under a CC-BY-4.0 license.
Please cite the Zenodo record if you use this dataset for your research.
1"""The FUGC dataset contains annotations for cervical lip segmentation in transvaginal ultrasound images. 2 3The dataset consists of 890 transvaginal ultrasound images (336x544 pixels, RGB) from the Fetal Ultrasound 4Grand Challenge on semi-supervised cervical segmentation. The anterior and posterior cervical lips were 5segmented automatically and then manually corrected by an experienced radiologist. 6 7Labeled images are available for three splits: 'train' (50), 'val' (90) and 'test' (300). The training split 8additionally contains 450 unlabeled images, which are not used by this loader. The label ids are 0 (background), 91 (anterior lip) and 2 (posterior lip). The record only names the two lips without listing the ids, so this 10order is inferred: the structure with id 1 lies above the one with id 2 in all 140 labeled training and 11validation images. 12 13The data is located at https://doi.org/10.5281/zenodo.16893174, released under a CC-BY-4.0 license. 14 15Please cite the Zenodo record if you use this dataset for your research. 16""" 17 18import os 19from glob import glob 20from natsort import natsorted 21from typing import Union, Tuple, Literal, List 22 23from torch.utils.data import Dataset, DataLoader 24 25import torch_em 26 27from .. import util 28 29 30URL = "https://zenodo.org/records/16893174/files/FUGC%20(Dataset).zip" 31CHECKSUM = "e4dea63dd744885835c9ce8035a640a6ae42780ae21180601ddb09590fa73ce4" 32 33SPLITS = {"train": os.path.join("train", "labeled_data"), "val": "val", "test": "test"} 34 35 36def get_fugc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 37 """Download the FUGC dataset. 38 39 Args: 40 path: Filepath to a folder where the data is downloaded for further processing. 41 download: Whether to download the data if it is not present. 42 43 Returns: 44 Filepath where the data is downloaded. 45 """ 46 data_dir = os.path.join(path, "FUGC (Dataset)", "dataset") 47 if os.path.exists(data_dir): 48 return data_dir 49 50 os.makedirs(path, exist_ok=True) 51 52 zip_path = os.path.join(path, "FUGC.zip") 53 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 54 util.unzip(zip_path=zip_path, dst=path) 55 56 assert os.path.exists(data_dir), f"The extraction of the FUGC archive did not create '{data_dir}'." 57 58 return data_dir 59 60 61def get_fugc_paths( 62 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False, 63) -> Tuple[List[str], List[str]]: 64 """Get paths to the FUGC data. 65 66 Args: 67 path: Filepath to a folder where the data is downloaded for further processing. 68 split: The choice of data split. One of 'train', 'val' or 'test'. 69 download: Whether to download the data if it is not present. 70 71 Returns: 72 List of filepaths for the image data. 73 List of filepaths for the label data. 74 """ 75 if split not in SPLITS: 76 raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.") 77 78 data_dir = get_fugc_data(path, download) 79 80 label_paths = natsorted(glob(os.path.join(data_dir, SPLITS[split], "labels", "*.png"))) 81 raw_paths = [os.path.join(data_dir, SPLITS[split], "images", os.path.basename(p)) for p in label_paths] 82 83 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 84 assert all(os.path.exists(p) for p in raw_paths) 85 86 return raw_paths, label_paths 87 88 89def get_fugc_dataset( 90 path: Union[os.PathLike, str], 91 patch_shape: Tuple[int, int], 92 split: Literal["train", "val", "test"], 93 resize_inputs: bool = False, 94 download: bool = False, 95 **kwargs 96) -> Dataset: 97 """Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound. 98 99 Args: 100 path: Filepath to a folder where the data is downloaded for further processing. 101 patch_shape: The patch shape to use for training. 102 split: The choice of data split. One of 'train', 'val' or 'test'. 103 resize_inputs: Whether to resize the inputs to the patch shape. 104 download: Whether to download the data if it is not present. 105 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 106 107 Returns: 108 The segmentation dataset. 109 """ 110 raw_paths, label_paths = get_fugc_paths(path, split, download) 111 112 if resize_inputs: 113 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 114 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 115 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 116 ) 117 118 return torch_em.default_segmentation_dataset( 119 raw_paths=raw_paths, 120 raw_key=None, 121 label_paths=label_paths, 122 label_key=None, 123 is_seg_dataset=False, 124 patch_shape=patch_shape, 125 **kwargs 126 ) 127 128 129def get_fugc_loader( 130 path: Union[os.PathLike, str], 131 batch_size: int, 132 patch_shape: Tuple[int, int], 133 split: Literal["train", "val", "test"], 134 resize_inputs: bool = False, 135 download: bool = False, 136 **kwargs 137) -> DataLoader: 138 """Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound. 139 140 Args: 141 path: Filepath to a folder where the data is downloaded for further processing. 142 batch_size: The batch size for training. 143 patch_shape: The patch shape to use for training. 144 split: The choice of data split. One of 'train', 'val' or 'test'. 145 resize_inputs: Whether to resize the inputs to the patch shape. 146 download: Whether to download the data if it is not present. 147 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 148 149 Returns: 150 The DataLoader. 151 """ 152 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 153 dataset = get_fugc_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 154 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
37def get_fugc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 38 """Download the FUGC dataset. 39 40 Args: 41 path: Filepath to a folder where the data is downloaded for further processing. 42 download: Whether to download the data if it is not present. 43 44 Returns: 45 Filepath where the data is downloaded. 46 """ 47 data_dir = os.path.join(path, "FUGC (Dataset)", "dataset") 48 if os.path.exists(data_dir): 49 return data_dir 50 51 os.makedirs(path, exist_ok=True) 52 53 zip_path = os.path.join(path, "FUGC.zip") 54 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 55 util.unzip(zip_path=zip_path, dst=path) 56 57 assert os.path.exists(data_dir), f"The extraction of the FUGC archive did not create '{data_dir}'." 58 59 return data_dir
Download the FUGC dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
62def get_fugc_paths( 63 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False, 64) -> Tuple[List[str], List[str]]: 65 """Get paths to the FUGC data. 66 67 Args: 68 path: Filepath to a folder where the data is downloaded for further processing. 69 split: The choice of data split. One of 'train', 'val' or 'test'. 70 download: Whether to download the data if it is not present. 71 72 Returns: 73 List of filepaths for the image data. 74 List of filepaths for the label data. 75 """ 76 if split not in SPLITS: 77 raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.") 78 79 data_dir = get_fugc_data(path, download) 80 81 label_paths = natsorted(glob(os.path.join(data_dir, SPLITS[split], "labels", "*.png"))) 82 raw_paths = [os.path.join(data_dir, SPLITS[split], "images", os.path.basename(p)) for p in label_paths] 83 84 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 85 assert all(os.path.exists(p) for p in raw_paths) 86 87 return raw_paths, label_paths
Get paths to the FUGC data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. One of 'train', 'val' or 'test'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
90def get_fugc_dataset( 91 path: Union[os.PathLike, str], 92 patch_shape: Tuple[int, int], 93 split: Literal["train", "val", "test"], 94 resize_inputs: bool = False, 95 download: bool = False, 96 **kwargs 97) -> Dataset: 98 """Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound. 99 100 Args: 101 path: Filepath to a folder where the data is downloaded for further processing. 102 patch_shape: The patch shape to use for training. 103 split: The choice of data split. One of 'train', 'val' or 'test'. 104 resize_inputs: Whether to resize the inputs to the patch shape. 105 download: Whether to download the data if it is not present. 106 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 107 108 Returns: 109 The segmentation dataset. 110 """ 111 raw_paths, label_paths = get_fugc_paths(path, split, download) 112 113 if resize_inputs: 114 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 115 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 116 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 117 ) 118 119 return torch_em.default_segmentation_dataset( 120 raw_paths=raw_paths, 121 raw_key=None, 122 label_paths=label_paths, 123 label_key=None, 124 is_seg_dataset=False, 125 patch_shape=patch_shape, 126 **kwargs 127 )
Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. One of 'train', 'val' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
130def get_fugc_loader( 131 path: Union[os.PathLike, str], 132 batch_size: int, 133 patch_shape: Tuple[int, int], 134 split: Literal["train", "val", "test"], 135 resize_inputs: bool = False, 136 download: bool = False, 137 **kwargs 138) -> DataLoader: 139 """Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound. 140 141 Args: 142 path: Filepath to a folder where the data is downloaded for further processing. 143 batch_size: The batch size for training. 144 patch_shape: The patch shape to use for training. 145 split: The choice of data split. One of 'train', 'val' or 'test'. 146 resize_inputs: Whether to resize the inputs to the patch shape. 147 download: Whether to download the data if it is not present. 148 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 149 150 Returns: 151 The DataLoader. 152 """ 153 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 154 dataset = get_fugc_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 155 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. One of 'train', 'val' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.