torch_em.data.datasets.medical.us_sim_and_seg
The US Simulation & Segmentation dataset contains real and simulated abdominal ultrasound scans with manual segmentations of several abdominal organs.
The real scans ('rus') were acquired from 11 healthy subjects with a SonoSite M Turbo V1.3 ultrasound device. The simulated scans ('aus') were generated with a ray-casting based simulator from CT volumes of the VISCERAL Anatomy3 challenge. Only a subset of the images in each domain has manual (real) or silver-standard (simulated) annotations of the liver, kidney, pancreas, vessels, adrenals, gallbladder, bones and spleen.
The dataset is located at https://www.kaggle.com/datasets/ignaciorlando/ussimandsegm. This dataset is from the publication https://doi.org/10.1007/s11548-019-02046-5. Please cite it if you use this dataset for your research.
1"""The US Simulation & Segmentation dataset contains real and simulated abdominal ultrasound 2scans with manual segmentations of several abdominal organs. 3 4The real scans ('rus') were acquired from 11 healthy subjects with a SonoSite M Turbo V1.3 5ultrasound device. The simulated scans ('aus') were generated with a ray-casting based 6simulator from CT volumes of the VISCERAL Anatomy3 challenge. Only a subset of the images 7in each domain has manual (real) or silver-standard (simulated) annotations of the liver, 8kidney, pancreas, vessels, adrenals, gallbladder, bones and spleen. 9 10The dataset is located at https://www.kaggle.com/datasets/ignaciorlando/ussimandsegm. 11This dataset is from the publication https://doi.org/10.1007/s11548-019-02046-5. 12Please cite it if you use this dataset for your research. 13""" 14 15import os 16from glob import glob 17from tqdm import tqdm 18from pathlib import Path 19from natsort import natsorted 20from typing import Union, Tuple, Literal, List 21 22import numpy as np 23import imageio.v3 as imageio 24 25from torch.utils.data import Dataset, DataLoader 26 27import torch_em 28 29from .. import util 30from ..light_microscopy.neurips_cell_seg import to_rgb 31 32 33LABEL_MAPS = { 34 (0, 0, 0): 0, # background (outside the field of view) 35 (10, 10, 10): 0, # background tissue (only present in the simulated annotations) 36 (100, 0, 100): 1, # liver (violet) 37 (255, 255, 0): 2, # kidney (yellow) 38 (0, 0, 255): 3, # pancreas (blue) 39 (255, 0, 0): 4, # vessels (red) 40 (0, 255, 255): 5, # adrenals (light blue) 41 (0, 255, 0): 6, # gallbladder (green) 42 (255, 255, 255): 7, # bones (white) 43 (255, 0, 255): 8, # spleen (pink) 44} 45 46 47def get_us_sim_and_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str: 48 """Download the US Simulation & Segmentation dataset. 49 50 Args: 51 path: Filepath to a folder where the data is downloaded for further processing. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 data_dir = os.path.join(path, "abdominal_US", "abdominal_US") 58 if os.path.exists(data_dir): 59 return data_dir 60 61 os.makedirs(path, exist_ok=True) 62 63 zip_path = os.path.join(path, "ussimandsegm.zip") 64 util.download_source_kaggle(path=path, dataset_name="ignaciorlando/ussimandsegm", download=download) 65 util.unzip(zip_path=zip_path, dst=path) 66 67 return data_dir 68 69 70def _preprocess_labels(image_paths, gt_paths, gt_dir): 71 os.makedirs(gt_dir, exist_ok=True) 72 73 neu_gt_paths = [] 74 for image_path, gt_path in tqdm( 75 zip(image_paths, gt_paths), total=len(image_paths), desc="Preprocessing labels" 76 ): 77 neu_gt_path = os.path.join(gt_dir, f"{Path(image_path).stem}.tif") 78 neu_gt_paths.append(neu_gt_path) 79 if os.path.exists(neu_gt_path): 80 continue 81 82 gt = imageio.imread(gt_path) 83 if gt.ndim == 2: 84 gt = np.stack([gt] * 3, axis=-1) 85 gt = gt[..., :3] # some annotations have an additional (constant) alpha channel. 86 87 labels = np.zeros(gt.shape[:2], dtype="uint8") 88 for color, label_id in LABEL_MAPS.items(): 89 if label_id == 0: 90 continue 91 binary_map = (gt == color).all(axis=-1) 92 labels[binary_map] = label_id 93 94 imageio.imwrite(neu_gt_path, labels, compression="zlib") 95 96 return neu_gt_paths 97 98 99def get_us_sim_and_seg_paths( 100 path: Union[os.PathLike, str], 101 source: Literal["aus", "rus"], 102 split: Literal["train", "test"], 103 download: bool = False, 104) -> Tuple[List[str], List[str]]: 105 """Get paths to the US Simulation & Segmentation data. 106 107 Args: 108 path: Filepath to a folder where the data is downloaded for further processing. 109 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 110 split: The choice of data split. Either 'train' or 'test'. 111 download: Whether to download the data if it is not present. 112 113 Returns: 114 List of filepaths for the image data. 115 List of filepaths for the label data. 116 """ 117 if source not in ("aus", "rus"): 118 raise ValueError(f"'{source}' is not a valid source. Choose either 'aus' or 'rus'.") 119 120 if split not in ("train", "test"): 121 raise ValueError(f"'{split}' is not a valid split. Choose either 'train' or 'test'.") 122 123 data_dir = get_us_sim_and_seg_data(path, download) 124 125 image_dir = os.path.join(data_dir, source.upper(), "images", split) 126 gt_dir = os.path.join(data_dir, source.upper(), "annotations", split) 127 128 if not os.path.exists(gt_dir): 129 raise ValueError(f"There are no annotations for the '{source}' source and the '{split}' split.") 130 131 image_paths = natsorted(glob(os.path.join(image_dir, "*"))) 132 gt_paths = natsorted(glob(os.path.join(gt_dir, "*"))) 133 134 image_stems = {Path(p).stem: p for p in image_paths} 135 gt_stems = {Path(p).stem: p for p in gt_paths} 136 matched_stems = natsorted(set(image_stems) & set(gt_stems)) 137 assert len(matched_stems) > 0 138 139 image_paths = [image_stems[stem] for stem in matched_stems] 140 gt_paths = [gt_stems[stem] for stem in matched_stems] 141 142 neu_gt_dir = os.path.join(data_dir, source.upper(), "preprocessed", split) 143 neu_gt_paths = _preprocess_labels(image_paths, gt_paths, neu_gt_dir) 144 145 return image_paths, neu_gt_paths 146 147 148def get_us_sim_and_seg_dataset( 149 path: Union[os.PathLike, str], 150 patch_shape: Tuple[int, int], 151 source: Literal["aus", "rus"], 152 split: Literal["train", "test"], 153 resize_inputs: bool = False, 154 download: bool = False, 155 **kwargs 156) -> Dataset: 157 """Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images. 158 159 Args: 160 path: Filepath to a folder where the data is downloaded for further processing. 161 patch_shape: The patch shape to use for training. 162 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 163 split: The choice of data split. Either 'train' or 'test'. 164 resize_inputs: Whether to resize the inputs to the patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 167 168 Returns: 169 The segmentation dataset. 170 """ 171 image_paths, gt_paths = get_us_sim_and_seg_paths(path, source, split, download) 172 173 if resize_inputs: 174 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 175 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 176 kwargs=kwargs, 177 patch_shape=patch_shape, 178 resize_inputs=resize_inputs, 179 resize_kwargs=resize_kwargs, 180 ensure_rgb=to_rgb, 181 ) 182 183 return torch_em.default_segmentation_dataset( 184 raw_paths=image_paths, 185 raw_key=None, 186 label_paths=gt_paths, 187 label_key=None, 188 patch_shape=patch_shape, 189 is_seg_dataset=False, 190 **kwargs 191 ) 192 193 194def get_us_sim_and_seg_loader( 195 path: Union[os.PathLike, str], 196 batch_size: int, 197 patch_shape: Tuple[int, int], 198 source: Literal["aus", "rus"], 199 split: Literal["train", "test"], 200 resize_inputs: bool = False, 201 download: bool = False, 202 **kwargs 203) -> DataLoader: 204 """Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images. 205 206 Args: 207 path: Filepath to a folder where the data is downloaded for further processing. 208 batch_size: The batch size for training. 209 patch_shape: The patch shape to use for training. 210 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 211 split: The choice of data split. Either 'train' or 'test'. 212 resize_inputs: Whether to resize the inputs to the patch shape. 213 download: Whether to download the data if it is not present. 214 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 215 216 Returns: 217 The DataLoader. 218 """ 219 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 220 dataset = get_us_sim_and_seg_dataset(path, patch_shape, source, split, resize_inputs, download, **ds_kwargs) 221 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
48def get_us_sim_and_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str: 49 """Download the US Simulation & Segmentation dataset. 50 51 Args: 52 path: Filepath to a folder where the data is downloaded for further processing. 53 download: Whether to download the data if it is not present. 54 55 Returns: 56 Filepath where the data is downloaded. 57 """ 58 data_dir = os.path.join(path, "abdominal_US", "abdominal_US") 59 if os.path.exists(data_dir): 60 return data_dir 61 62 os.makedirs(path, exist_ok=True) 63 64 zip_path = os.path.join(path, "ussimandsegm.zip") 65 util.download_source_kaggle(path=path, dataset_name="ignaciorlando/ussimandsegm", download=download) 66 util.unzip(zip_path=zip_path, dst=path) 67 68 return data_dir
Download the US Simulation & Segmentation dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
100def get_us_sim_and_seg_paths( 101 path: Union[os.PathLike, str], 102 source: Literal["aus", "rus"], 103 split: Literal["train", "test"], 104 download: bool = False, 105) -> Tuple[List[str], List[str]]: 106 """Get paths to the US Simulation & Segmentation data. 107 108 Args: 109 path: Filepath to a folder where the data is downloaded for further processing. 110 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 111 split: The choice of data split. Either 'train' or 'test'. 112 download: Whether to download the data if it is not present. 113 114 Returns: 115 List of filepaths for the image data. 116 List of filepaths for the label data. 117 """ 118 if source not in ("aus", "rus"): 119 raise ValueError(f"'{source}' is not a valid source. Choose either 'aus' or 'rus'.") 120 121 if split not in ("train", "test"): 122 raise ValueError(f"'{split}' is not a valid split. Choose either 'train' or 'test'.") 123 124 data_dir = get_us_sim_and_seg_data(path, download) 125 126 image_dir = os.path.join(data_dir, source.upper(), "images", split) 127 gt_dir = os.path.join(data_dir, source.upper(), "annotations", split) 128 129 if not os.path.exists(gt_dir): 130 raise ValueError(f"There are no annotations for the '{source}' source and the '{split}' split.") 131 132 image_paths = natsorted(glob(os.path.join(image_dir, "*"))) 133 gt_paths = natsorted(glob(os.path.join(gt_dir, "*"))) 134 135 image_stems = {Path(p).stem: p for p in image_paths} 136 gt_stems = {Path(p).stem: p for p in gt_paths} 137 matched_stems = natsorted(set(image_stems) & set(gt_stems)) 138 assert len(matched_stems) > 0 139 140 image_paths = [image_stems[stem] for stem in matched_stems] 141 gt_paths = [gt_stems[stem] for stem in matched_stems] 142 143 neu_gt_dir = os.path.join(data_dir, source.upper(), "preprocessed", split) 144 neu_gt_paths = _preprocess_labels(image_paths, gt_paths, neu_gt_dir) 145 146 return image_paths, neu_gt_paths
Get paths to the US Simulation & Segmentation data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
- split: The choice of data split. Either 'train' or 'test'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
149def get_us_sim_and_seg_dataset( 150 path: Union[os.PathLike, str], 151 patch_shape: Tuple[int, int], 152 source: Literal["aus", "rus"], 153 split: Literal["train", "test"], 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> Dataset: 158 """Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 patch_shape: The patch shape to use for training. 163 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 164 split: The choice of data split. Either 'train' or 'test'. 165 resize_inputs: Whether to resize the inputs to the patch shape. 166 download: Whether to download the data if it is not present. 167 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 168 169 Returns: 170 The segmentation dataset. 171 """ 172 image_paths, gt_paths = get_us_sim_and_seg_paths(path, source, split, download) 173 174 if resize_inputs: 175 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 176 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 177 kwargs=kwargs, 178 patch_shape=patch_shape, 179 resize_inputs=resize_inputs, 180 resize_kwargs=resize_kwargs, 181 ensure_rgb=to_rgb, 182 ) 183 184 return torch_em.default_segmentation_dataset( 185 raw_paths=image_paths, 186 raw_key=None, 187 label_paths=gt_paths, 188 label_key=None, 189 patch_shape=patch_shape, 190 is_seg_dataset=False, 191 **kwargs 192 )
Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
- split: The choice of data split. Either 'train' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
195def get_us_sim_and_seg_loader( 196 path: Union[os.PathLike, str], 197 batch_size: int, 198 patch_shape: Tuple[int, int], 199 source: Literal["aus", "rus"], 200 split: Literal["train", "test"], 201 resize_inputs: bool = False, 202 download: bool = False, 203 **kwargs 204) -> DataLoader: 205 """Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images. 206 207 Args: 208 path: Filepath to a folder where the data is downloaded for further processing. 209 batch_size: The batch size for training. 210 patch_shape: The patch shape to use for training. 211 source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans). 212 split: The choice of data split. Either 'train' or 'test'. 213 resize_inputs: Whether to resize the inputs to the patch shape. 214 download: Whether to download the data if it is not present. 215 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 216 217 Returns: 218 The DataLoader. 219 """ 220 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 221 dataset = get_us_sim_and_seg_dataset(path, patch_shape, source, split, resize_inputs, download, **ds_kwargs) 222 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
- split: The choice of data split. Either 'train' or 'test'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.