torch_em.data.datasets.medical.ureteroscopy_lumen

The Ureteroscopy Lumen Segmentation dataset contains annotations for ureteral lumen segmentation in endoscopic ureteroscopy images.

The dataset provides 'train' and 'val' splits with images paired one-to-one with binary lumen masks, and a 'test' split organized per-patient ('test_01', 'test_02', 'test_03'), which this loader merges into a single flat split. The masks ship as near-binary RGB PNGs with a small amount of JPEG-style compression noise, so this loader binarizes them (foreground where any channel exceeds half intensity) and stores them as single-channel tif files during preprocessing.

NOTE: The Zenodo record description reports two inconsistent counts, "1,754 images from 23 patients" and "2,187 images with masks". This loader does not rely on either advertised count and instead discovers the image-mask pairs on disk, which totals 2,181 pairs (798 train, 417 val, 966 test).

The dataset is located at https://zenodo.org/records/10066606, released under a CC-BY-4.0 license.

This dataset is from the publication https://doi.org/10.1109/ICPR48806.2021.9412209. Please cite it if you use this dataset for your research.

  1"""The Ureteroscopy Lumen Segmentation dataset contains annotations for ureteral lumen segmentation
  2in endoscopic ureteroscopy images.
  3
  4The dataset provides 'train' and 'val' splits with images paired one-to-one with binary lumen masks,
  5and a 'test' split organized per-patient ('test_01', 'test_02', 'test_03'), which this loader merges
  6into a single flat split. The masks ship as near-binary RGB PNGs with a small amount of JPEG-style
  7compression noise, so this loader binarizes them (foreground where any channel exceeds half intensity)
  8and stores them as single-channel tif files during preprocessing.
  9
 10NOTE: The Zenodo record description reports two inconsistent counts, "1,754 images from 23 patients"
 11and "2,187 images with masks". This loader does not rely on either advertised count and instead
 12discovers the image-mask pairs on disk, which totals 2,181 pairs (798 train, 417 val, 966 test).
 13
 14The dataset is located at https://zenodo.org/records/10066606, released under a CC-BY-4.0 license.
 15
 16This dataset is from the publication https://doi.org/10.1109/ICPR48806.2021.9412209.
 17Please cite it if you use this dataset for your research.
 18"""
 19
 20import os
 21from glob import glob
 22from tqdm import tqdm
 23from natsort import natsorted
 24from typing import Union, Tuple, Literal, List
 25
 26import numpy as np
 27import imageio.v3 as imageio
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36URL = "https://zenodo.org/records/10066606/files/lumen_dataset.zip"
 37CHECKSUM = "e2a99c1bd59453d0bf7eed3fb916c0ea6526f4d9b65851d9759f919362effdc8"
 38
 39SPLITS = ["train", "val", "test"]
 40
 41
 42def _preprocess_data(data_dir):
 43    preprocessed_dir = os.path.join(data_dir, "preprocessed")
 44    if os.path.exists(preprocessed_dir):
 45        return preprocessed_dir
 46
 47    raw_dir = os.path.join(data_dir, "lumen_dataset")
 48    for split in SPLITS:
 49        image_dir = os.path.join(preprocessed_dir, split, "images")
 50        label_dir = os.path.join(preprocessed_dir, split, "labels")
 51        os.makedirs(image_dir, exist_ok=True)
 52        os.makedirs(label_dir, exist_ok=True)
 53
 54        if split == "test":
 55            source_dirs = natsorted(glob(os.path.join(raw_dir, "test", "test_*")))
 56        else:
 57            source_dirs = [os.path.join(raw_dir, split)]
 58
 59        for source_dir in tqdm(source_dirs, desc=f"Preprocessing '{split}' split"):
 60            label_paths = natsorted(glob(os.path.join(source_dir, "label", "*.png")))
 61            for label_path in label_paths:
 62                fname = os.path.basename(label_path)
 63                image_path = os.path.join(source_dir, "image", fname)
 64                if not os.path.exists(image_path):
 65                    continue
 66
 67                label = imageio.imread(label_path)
 68                label = (np.any(label > 127, axis=-1)).astype("uint8")
 69                image = imageio.imread(image_path)
 70
 71                out_name = os.path.splitext(f"{os.path.basename(source_dir)}_{fname}")[0] + ".tif"
 72                imageio.imwrite(os.path.join(image_dir, out_name), image)
 73                imageio.imwrite(os.path.join(label_dir, out_name), label)
 74
 75    return preprocessed_dir
 76
 77
 78def get_ureteroscopy_lumen_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 79    """Download the Ureteroscopy Lumen Segmentation dataset.
 80
 81    Args:
 82        path: Filepath to a folder where the data is downloaded for further processing.
 83        download: Whether to download the data if it is not present.
 84
 85    Returns:
 86        Filepath where the preprocessed data is stored.
 87    """
 88    data_dir = os.path.join(path, "lumen_dataset")
 89    if os.path.exists(data_dir):
 90        return _preprocess_data(path)
 91
 92    os.makedirs(path, exist_ok=True)
 93
 94    zip_path = os.path.join(path, "lumen_dataset.zip")
 95    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 96    util.unzip(zip_path=zip_path, dst=path)
 97
 98    return _preprocess_data(path)
 99
100
101def get_ureteroscopy_lumen_paths(
102    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False,
103) -> Tuple[List[str], List[str]]:
104    """Get paths to the Ureteroscopy Lumen Segmentation data.
105
106    Args:
107        path: Filepath to a folder where the data is downloaded for further processing.
108        split: The choice of data split. Either 'train', 'val' or 'test'.
109        download: Whether to download the data if it is not present.
110
111    Returns:
112        List of filepaths for the image data.
113        List of filepaths for the label data.
114    """
115    if split not in SPLITS:
116        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
117
118    preprocessed_dir = get_ureteroscopy_lumen_data(path, download)
119
120    raw_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "images", "*.tif")))
121    label_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "labels", "*.tif")))
122
123    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
124
125    return raw_paths, label_paths
126
127
128def get_ureteroscopy_lumen_dataset(
129    path: Union[os.PathLike, str],
130    patch_shape: Tuple[int, int],
131    split: Literal["train", "val", "test"],
132    resize_inputs: bool = False,
133    download: bool = False,
134    **kwargs
135) -> Dataset:
136    """Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation.
137
138    Args:
139        path: Filepath to a folder where the data is downloaded for further processing.
140        patch_shape: The patch shape to use for training.
141        split: The choice of data split. Either 'train', 'val' or 'test'.
142        resize_inputs: Whether to resize the inputs to the patch shape.
143        download: Whether to download the data if it is not present.
144        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
145
146    Returns:
147        The segmentation dataset.
148    """
149    raw_paths, label_paths = get_ureteroscopy_lumen_paths(path, split, download)
150
151    if resize_inputs:
152        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
153        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
154            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
155        )
156
157    return torch_em.default_segmentation_dataset(
158        raw_paths=raw_paths,
159        raw_key=None,
160        label_paths=label_paths,
161        label_key=None,
162        is_seg_dataset=False,
163        patch_shape=patch_shape,
164        **kwargs
165    )
166
167
168def get_ureteroscopy_lumen_loader(
169    path: Union[os.PathLike, str],
170    batch_size: int,
171    patch_shape: Tuple[int, int],
172    split: Literal["train", "val", "test"],
173    resize_inputs: bool = False,
174    download: bool = False,
175    **kwargs
176) -> DataLoader:
177    """Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation.
178
179    Args:
180        path: Filepath to a folder where the data is downloaded for further processing.
181        batch_size: The batch size for training.
182        patch_shape: The patch shape to use for training.
183        split: The choice of data split. Either 'train', 'val' or 'test'.
184        resize_inputs: Whether to resize the inputs to the patch shape.
185        download: Whether to download the data if it is not present.
186        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
187
188    Returns:
189        The DataLoader.
190    """
191    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
192    dataset = get_ureteroscopy_lumen_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
193    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/10066606/files/lumen_dataset.zip'
CHECKSUM = 'e2a99c1bd59453d0bf7eed3fb916c0ea6526f4d9b65851d9759f919362effdc8'
SPLITS = ['train', 'val', 'test']
def get_ureteroscopy_lumen_data(path: Union[os.PathLike, str], download: bool = False) -> str:
79def get_ureteroscopy_lumen_data(path: Union[os.PathLike, str], download: bool = False) -> str:
80    """Download the Ureteroscopy Lumen Segmentation dataset.
81
82    Args:
83        path: Filepath to a folder where the data is downloaded for further processing.
84        download: Whether to download the data if it is not present.
85
86    Returns:
87        Filepath where the preprocessed data is stored.
88    """
89    data_dir = os.path.join(path, "lumen_dataset")
90    if os.path.exists(data_dir):
91        return _preprocess_data(path)
92
93    os.makedirs(path, exist_ok=True)
94
95    zip_path = os.path.join(path, "lumen_dataset.zip")
96    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
97    util.unzip(zip_path=zip_path, dst=path)
98
99    return _preprocess_data(path)

Download the Ureteroscopy Lumen Segmentation dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_ureteroscopy_lumen_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> Tuple[List[str], List[str]]:
102def get_ureteroscopy_lumen_paths(
103    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False,
104) -> Tuple[List[str], List[str]]:
105    """Get paths to the Ureteroscopy Lumen Segmentation data.
106
107    Args:
108        path: Filepath to a folder where the data is downloaded for further processing.
109        split: The choice of data split. Either 'train', 'val' or 'test'.
110        download: Whether to download the data if it is not present.
111
112    Returns:
113        List of filepaths for the image data.
114        List of filepaths for the label data.
115    """
116    if split not in SPLITS:
117        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
118
119    preprocessed_dir = get_ureteroscopy_lumen_data(path, download)
120
121    raw_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "images", "*.tif")))
122    label_paths = natsorted(glob(os.path.join(preprocessed_dir, split, "labels", "*.tif")))
123
124    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
125
126    return raw_paths, label_paths

Get paths to the Ureteroscopy Lumen Segmentation data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_ureteroscopy_lumen_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
129def get_ureteroscopy_lumen_dataset(
130    path: Union[os.PathLike, str],
131    patch_shape: Tuple[int, int],
132    split: Literal["train", "val", "test"],
133    resize_inputs: bool = False,
134    download: bool = False,
135    **kwargs
136) -> Dataset:
137    """Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation.
138
139    Args:
140        path: Filepath to a folder where the data is downloaded for further processing.
141        patch_shape: The patch shape to use for training.
142        split: The choice of data split. Either 'train', 'val' or 'test'.
143        resize_inputs: Whether to resize the inputs to the patch shape.
144        download: Whether to download the data if it is not present.
145        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
146
147    Returns:
148        The segmentation dataset.
149    """
150    raw_paths, label_paths = get_ureteroscopy_lumen_paths(path, split, download)
151
152    if resize_inputs:
153        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
154        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
155            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
156        )
157
158    return torch_em.default_segmentation_dataset(
159        raw_paths=raw_paths,
160        raw_key=None,
161        label_paths=label_paths,
162        label_key=None,
163        is_seg_dataset=False,
164        patch_shape=patch_shape,
165        **kwargs
166    )

Get the Ureteroscopy Lumen Segmentation dataset for lumen segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_ureteroscopy_lumen_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
169def get_ureteroscopy_lumen_loader(
170    path: Union[os.PathLike, str],
171    batch_size: int,
172    patch_shape: Tuple[int, int],
173    split: Literal["train", "val", "test"],
174    resize_inputs: bool = False,
175    download: bool = False,
176    **kwargs
177) -> DataLoader:
178    """Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation.
179
180    Args:
181        path: Filepath to a folder where the data is downloaded for further processing.
182        batch_size: The batch size for training.
183        patch_shape: The patch shape to use for training.
184        split: The choice of data split. Either 'train', 'val' or 'test'.
185        resize_inputs: Whether to resize the inputs to the patch shape.
186        download: Whether to download the data if it is not present.
187        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
188
189    Returns:
190        The DataLoader.
191    """
192    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
193    dataset = get_ureteroscopy_lumen_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
194    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Ureteroscopy Lumen Segmentation dataloader for lumen segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. Either 'train', 'val' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.