torch_em.data.datasets.medical.ureteroscopy_lumen
The Ureteroscopy Lumen Segmentation dataset contains annotations for ureteral lumen segmentation in endoscopic ureteroscopy images.
The dataset provides 'train' and 'val' splits with images paired one-to-one with binary lumen masks, and a 'test' split organized per-patient ('test_01', 'test_02', 'test_03'), which this loader merges into a single flat split. The masks ship as near-binary RGB PNGs with a small amount of JPEG-style compression noise, so this loader binarizes them (foreground where any channel exceeds half intensity) and stores them as single-channel tif files during preprocessing.
NOTE: The Zenodo record description reports two inconsistent counts, "1,754 images from 23 patients" and "2,187 images with masks". This loader does not rely on either advertised count and instead discovers the image-mask pairs on disk, which totals 2,181 pairs (798 train, 417 val, 966 test).
The dataset is located at https://zenodo.org/records/10066606, released under a CC-BY-4.0 license.
This dataset is from the publication https://doi.org/10.1109/ICPR48806.2021.9412209. Please cite it if you use this dataset for your research.
1"""The Ureteroscopy Lumen Segmentation dataset contains annotations for ureteral lumen segmentation 2in endoscopic ureteroscopy images. 3 4The dataset provides 'train' and 'val' splits with images paired one-to-one with binary lumen masks, 5and a 'test' split organized per-patient ('test_01', 'test_02', 'test_03'), which this loader merges 6into a single flat split. The masks ship as near-binary RGB PNGs with a small amount of JPEG-style 7compression noise, so this loader binarizes them (foreground where any channel exceeds half intensity) 8and stores them as single-channel tif files during preprocessing. 9 10NOTE: The Zenodo record description reports two inconsistent counts, "1,754 images from 23 patients" 11and "2,187 images with masks". This loader does not rely on either advertised count and instead 12discovers the image-mask pairs on disk, which totals 2,181 pairs (798 train, 417 val, 966 test). 13 14The dataset is located at https://zenodo.org/records/10066606, released under a CC-BY-4.0 license. 15 16This dataset is from the publication https://doi.org/10.1109/ICPR48806.2021.9412209. 17Please cite it if you use this dataset for your research. 18""" 19 20import os 21from glob import glob 22from tqdm import tqdm 23from natsort import natsorted 24from typing import Union, Tuple, Literal, List 25 26import numpy as np 27import imageio.v3 as imageio 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36URL = "https://zenodo.org/records/10066606/files/lumen_dataset.zip" 37CHECKSUM = "e2a99c1bd59453d0bf7eed3fb916c0ea6526f4d9b65851d9759f919362effdc8" 38 39SPLITS = ["train", "val", "test"] 40 41 42def _preprocess_data(data_dir): 43 preprocessed_dir = os.path.join(data_dir, "preprocessed") 44 if os.path.exists(preprocessed_dir): 45 return preprocessed_dir 46 47 raw_dir = os.path.join(data_dir, "lumen_dataset") 48 for split in SPLITS: 49 image_dir = os.path.join(preprocessed_dir, split, "images") 50 label_dir = os.path.join(preprocessed_dir, split, "labels") 51 os.makedirs(image_dir, exist_ok=True) 52 os.makedirs(label_dir, exist_ok=True) 53 54 if split == "test": 55 source_dirs = natsorted(glob(os.path.join(raw_dir, "test", "test_*"))) 56 else: 57 source_dirs = [os.path.join(raw_dir, split)] 58 59 for source_dir in tqdm(source_dirs, desc=f"Preprocessing '{split}' split"): 60 label_paths = natsorted(glob(os.path.join(source_dir, "label", "*.png"))) 61 for label_path in label_paths: 62 fname = os.path.basename(label_path) 63 image_path = os.path.join(source_dir, "image", fname) 64 if not os.path.exists(image_path): 65 continue 66 67 label = imageio.imread(label_path) 68 label = (np.any(label > 127, axis=-1)).astype("uint8") 69 image = imageio.imread(image_path) 70 71 out_name = os.path.splitext(f"{os.path.basename(source_dir)}_{fname}")[0] + ".tif" 72 imageio.imwrite(os.path.join(image_dir, out_name), image) 73 imageio.imwrite(os.path.join(label_dir, out_name), label) 74 75 return preprocessed_dir 76 77 78def get_ureteroscopy_lumen_data(path: Union[os.PathLike, str], download: bool = False) -> str: 79 """Download the Ureteroscopy Lumen Segmentation dataset. 80 81 Args: 82 path: Filepath to a folder where the data is downloaded for further processing. 83 download: Whether to download the data if it is not present. 84 85 Returns: 86 Filepath where the preprocessed data is stored. 87 """ 88 data_dir = os.path.join(path, "lumen_dataset") 89 if os.path.exists(data_dir): 90 return _preprocess_data(path) 91 92 os.makedirs(path, exist_ok=True) 93 94 zip_path = os.path.join(path, "lumen_dataset.zip") 95 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 96 util.unzip(zip_path=zip_path, dst=path) 97 98 return _preprocess_data(path) 99 100 101def get_ureteroscopy_lumen_paths( 102 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False, 103) -> Tuple[List[str], List[str]]: 104 """Get paths to the Ureteroscopy Lumen Segmentation data. 105 106 Args: 107 path: Filepath to a folder where the data is downloaded for further processing. 108 split: The choice of data split. Either 'train', 'val' or 'test'. 109 download: Whether to download the data if it is not present. 110 111 Returns: 112 List of filepaths for the image data. 113 List of filepaths for the label data. 114 """ 115 if split not in SPLITS: 116 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 117 118 preprocessed_dir = get_ureteroscopy_lumen_data(path, download) 119 120 raw_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "images", "*.tif"))) 121 label_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "labels", "*.tif"))) 122 123 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 124 125 return raw_paths, label_paths 126 127 128def get_ureteroscopy_lumen_dataset( 129 path: Union[os.PathLike, str], 130 patch_shape: Tuple[int, int], 131 split: Literal["train", "val", "test"], 132 resize_inputs: bool = False, 133 download: bool = False, 134 **kwargs 135) -> Dataset: 136 """Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation. 137 138 Args: 139 path: Filepath to a folder where the data is downloaded for further processing. 140 patch_shape: The patch shape to use for training. 141 split: The choice of data split. Either 'train', 'val' or 'test'. 142 resize_inputs: Whether to resize the inputs to the patch shape. 143 download: Whether to download the data if it is not present. 144 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 145 146 Returns: 147 The segmentation dataset. 148 """ 149 raw_paths, label_paths = get_ureteroscopy_lumen_paths(path, split, download) 150 151 if resize_inputs: 152 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 153 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 154 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 155 ) 156 157 return torch_em.default_segmentation_dataset( 158 raw_paths=raw_paths, 159 raw_key=None, 160 label_paths=label_paths, 161 label_key=None, 162 is_seg_dataset=False, 163 patch_shape=patch_shape, 164 **kwargs 165 ) 166 167 168def get_ureteroscopy_lumen_loader( 169 path: Union[os.PathLike, str], 170 batch_size: int, 171 patch_shape: Tuple[int, int], 172 split: Literal["train", "val", "test"], 173 resize_inputs: bool = False, 174 download: bool = False, 175 **kwargs 176) -> DataLoader: 177 """Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation. 178 179 Args: 180 path: Filepath to a folder where the data is downloaded for further processing. 181 batch_size: The batch size for training. 182 patch_shape: The patch shape to use for training. 183 split: The choice of data split. Either 'train', 'val' or 'test'. 184 resize_inputs: Whether to resize the inputs to the patch shape. 185 download: Whether to download the data if it is not present. 186 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 187 188 Returns: 189 The DataLoader. 190 """ 191 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 192 dataset = get_ureteroscopy_lumen_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 193 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
79def get_ureteroscopy_lumen_data(path: Union[os.PathLike, str], download: bool = False) -> str: 80 """Download the Ureteroscopy Lumen Segmentation dataset. 81 82 Args: 83 path: Filepath to a folder where the data is downloaded for further processing. 84 download: Whether to download the data if it is not present. 85 86 Returns: 87 Filepath where the preprocessed data is stored. 88 """ 89 data_dir = os.path.join(path, "lumen_dataset") 90 if os.path.exists(data_dir): 91 return _preprocess_data(path) 92 93 os.makedirs(path, exist_ok=True) 94 95 zip_path = os.path.join(path, "lumen_dataset.zip") 96 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 97 util.unzip(zip_path=zip_path, dst=path) 98 99 return _preprocess_data(path)
Download the Ureteroscopy Lumen Segmentation dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
102def get_ureteroscopy_lumen_paths( 103 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False, 104) -> Tuple[List[str], List[str]]: 105 """Get paths to the Ureteroscopy Lumen Segmentation data. 106 107 Args: 108 path: Filepath to a folder where the data is downloaded for further processing. 109 split: The choice of data split. Either 'train', 'val' or 'test'. 110 download: Whether to download the data if it is not present. 111 112 Returns: 113 List of filepaths for the image data. 114 List of filepaths for the label data. 115 """ 116 if split not in SPLITS: 117 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 118 119 preprocessed_dir = get_ureteroscopy_lumen_data(path, download) 120 121 raw_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "images", "*.tif"))) 122 label_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "labels", "*.tif"))) 123 124 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 125 126 return raw_paths, label_paths
Get paths to the Ureteroscopy Lumen Segmentation data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
129def get_ureteroscopy_lumen_dataset( 130 path: Union[os.PathLike, str], 131 patch_shape: Tuple[int, int], 132 split: Literal["train", "val", "test"], 133 resize_inputs: bool = False, 134 download: bool = False, 135 **kwargs 136) -> Dataset: 137 """Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation. 138 139 Args: 140 path: Filepath to a folder where the data is downloaded for further processing. 141 patch_shape: The patch shape to use for training. 142 split: The choice of data split. Either 'train', 'val' or 'test'. 143 resize_inputs: Whether to resize the inputs to the patch shape. 144 download: Whether to download the data if it is not present. 145 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 146 147 Returns: 148 The segmentation dataset. 149 """ 150 raw_paths, label_paths = get_ureteroscopy_lumen_paths(path, split, download) 151 152 if resize_inputs: 153 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 154 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 155 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 156 ) 157 158 return torch_em.default_segmentation_dataset( 159 raw_paths=raw_paths, 160 raw_key=None, 161 label_paths=label_paths, 162 label_key=None, 163 is_seg_dataset=False, 164 patch_shape=patch_shape, 165 **kwargs 166 )
Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
169def get_ureteroscopy_lumen_loader( 170 path: Union[os.PathLike, str], 171 batch_size: int, 172 patch_shape: Tuple[int, int], 173 split: Literal["train", "val", "test"], 174 resize_inputs: bool = False, 175 download: bool = False, 176 **kwargs 177) -> DataLoader: 178 """Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation. 179 180 Args: 181 path: Filepath to a folder where the data is downloaded for further processing. 182 batch_size: The batch size for training. 183 patch_shape: The patch shape to use for training. 184 split: The choice of data split. Either 'train', 'val' or 'test'. 185 resize_inputs: Whether to resize the inputs to the patch shape. 186 download: Whether to download the data if it is not present. 187 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 188 189 Returns: 190 The DataLoader. 191 """ 192 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 193 dataset = get_ureteroscopy_lumen_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 194 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train', 'val' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.