torch_em.data.datasets.medical.us_sim_and_seg

The US Simulation & Segmentation dataset contains real and simulated abdominal ultrasound scans with manual segmentations of several abdominal organs.

The real scans ('rus') were acquired from 11 healthy subjects with a SonoSite M Turbo V1.3 ultrasound device. The simulated scans ('aus') were generated with a ray-casting based simulator from CT volumes of the VISCERAL Anatomy3 challenge. Only a subset of the images in each domain has manual (real) or silver-standard (simulated) annotations of the liver, kidney, pancreas, vessels, adrenals, gallbladder, bones and spleen.

The dataset is located at https://www.kaggle.com/datasets/ignaciorlando/ussimandsegm. This dataset is from the publication https://doi.org/10.1007/s11548-019-02046-5. Please cite it if you use this dataset for your research.

  1"""The US Simulation & Segmentation dataset contains real and simulated abdominal ultrasound
  2scans with manual segmentations of several abdominal organs.
  3
  4The real scans ('rus') were acquired from 11 healthy subjects with a SonoSite M Turbo V1.3
  5ultrasound device. The simulated scans ('aus') were generated with a ray-casting based
  6simulator from CT volumes of the VISCERAL Anatomy3 challenge. Only a subset of the images
  7in each domain has manual (real) or silver-standard (simulated) annotations of the liver,
  8kidney, pancreas, vessels, adrenals, gallbladder, bones and spleen.
  9
 10The dataset is located at https://www.kaggle.com/datasets/ignaciorlando/ussimandsegm.
 11This dataset is from the publication https://doi.org/10.1007/s11548-019-02046-5.
 12Please cite it if you use this dataset for your research.
 13"""
 14
 15import os
 16from glob import glob
 17from tqdm import tqdm
 18from pathlib import Path
 19from natsort import natsorted
 20from typing import Union, Tuple, Literal, List
 21
 22import numpy as np
 23import imageio.v3 as imageio
 24
 25from torch.utils.data import Dataset, DataLoader
 26
 27import torch_em
 28
 29from .. import util
 30from ..light_microscopy.neurips_cell_seg import to_rgb
 31
 32
 33LABEL_MAPS = {
 34    (0, 0, 0): 0,  # background (outside the field of view)
 35    (10, 10, 10): 0,  # background tissue (only present in the simulated annotations)
 36    (100, 0, 100): 1,  # liver (violet)
 37    (255, 255, 0): 2,  # kidney (yellow)
 38    (0, 0, 255): 3,  # pancreas (blue)
 39    (255, 0, 0): 4,  # vessels (red)
 40    (0, 255, 255): 5,  # adrenals (light blue)
 41    (0, 255, 0): 6,  # gallbladder (green)
 42    (255, 255, 255): 7,  # bones (white)
 43    (255, 0, 255): 8,  # spleen (pink)
 44}
 45
 46
 47def get_us_sim_and_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 48    """Download the US Simulation & Segmentation dataset.
 49
 50    Args:
 51        path: Filepath to a folder where the data is downloaded for further processing.
 52        download: Whether to download the data if it is not present.
 53
 54    Returns:
 55        Filepath where the data is downloaded.
 56    """
 57    data_dir = os.path.join(path, "abdominal_US", "abdominal_US")
 58    if os.path.exists(data_dir):
 59        return data_dir
 60
 61    os.makedirs(path, exist_ok=True)
 62
 63    zip_path = os.path.join(path, "ussimandsegm.zip")
 64    util.download_source_kaggle(path=path, dataset_name="ignaciorlando/ussimandsegm", download=download)
 65    util.unzip(zip_path=zip_path, dst=path)
 66
 67    return data_dir
 68
 69
 70def _preprocess_labels(image_paths, gt_paths, gt_dir):
 71    os.makedirs(gt_dir, exist_ok=True)
 72
 73    neu_gt_paths = []
 74    for image_path, gt_path in tqdm(
 75        zip(image_paths, gt_paths), total=len(image_paths), desc="Preprocessing labels"
 76    ):
 77        neu_gt_path = os.path.join(gt_dir, f"{Path(image_path).stem}.tif")
 78        neu_gt_paths.append(neu_gt_path)
 79        if os.path.exists(neu_gt_path):
 80            continue
 81
 82        gt = imageio.imread(gt_path)
 83        if gt.ndim == 2:
 84            gt = np.stack([gt] * 3, axis=-1)
 85        gt = gt[..., :3]  # some annotations have an additional (constant) alpha channel.
 86
 87        labels = np.zeros(gt.shape[:2], dtype="uint8")
 88        for color, label_id in LABEL_MAPS.items():
 89            if label_id == 0:
 90                continue
 91            binary_map = (gt == color).all(axis=-1)
 92            labels[binary_map] = label_id
 93
 94        imageio.imwrite(neu_gt_path, labels, compression="zlib")
 95
 96    return neu_gt_paths
 97
 98
 99def get_us_sim_and_seg_paths(
100    path: Union[os.PathLike, str],
101    source: Literal["aus", "rus"],
102    split: Literal["train", "test"],
103    download: bool = False,
104) -> Tuple[List[str], List[str]]:
105    """Get paths to the US Simulation & Segmentation data.
106
107    Args:
108        path: Filepath to a folder where the data is downloaded for further processing.
109        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
110        split: The choice of data split. Either 'train' or 'test'.
111        download: Whether to download the data if it is not present.
112
113    Returns:
114        List of filepaths for the image data.
115        List of filepaths for the label data.
116    """
117    if source not in ("aus", "rus"):
118        raise ValueError(f"'{source}' is not a valid source. Choose either 'aus' or 'rus'.")
119
120    if split not in ("train", "test"):
121        raise ValueError(f"'{split}' is not a valid split. Choose either 'train' or 'test'.")
122
123    data_dir = get_us_sim_and_seg_data(path, download)
124
125    image_dir = os.path.join(data_dir, source.upper(), "images", split)
126    gt_dir = os.path.join(data_dir, source.upper(), "annotations", split)
127
128    if not os.path.exists(gt_dir):
129        raise ValueError(f"There are no annotations for the '{source}' source and the '{split}' split.")
130
131    image_paths = natsorted(glob(os.path.join(image_dir, "*")))
132    gt_paths = natsorted(glob(os.path.join(gt_dir, "*")))
133
134    image_stems = {Path(p).stem: p for p in image_paths}
135    gt_stems = {Path(p).stem: p for p in gt_paths}
136    matched_stems = natsorted(set(image_stems) & set(gt_stems))
137    assert len(matched_stems) > 0
138
139    image_paths = [image_stems[stem] for stem in matched_stems]
140    gt_paths = [gt_stems[stem] for stem in matched_stems]
141
142    neu_gt_dir = os.path.join(data_dir, source.upper(), "preprocessed", split)
143    neu_gt_paths = _preprocess_labels(image_paths, gt_paths, neu_gt_dir)
144
145    return image_paths, neu_gt_paths
146
147
148def get_us_sim_and_seg_dataset(
149    path: Union[os.PathLike, str],
150    patch_shape: Tuple[int, int],
151    source: Literal["aus", "rus"],
152    split: Literal["train", "test"],
153    resize_inputs: bool = False,
154    download: bool = False,
155    **kwargs
156) -> Dataset:
157    """Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images.
158
159    Args:
160        path: Filepath to a folder where the data is downloaded for further processing.
161        patch_shape: The patch shape to use for training.
162        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
163        split: The choice of data split. Either 'train' or 'test'.
164        resize_inputs: Whether to resize the inputs to the patch shape.
165        download: Whether to download the data if it is not present.
166        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
167
168    Returns:
169        The segmentation dataset.
170    """
171    image_paths, gt_paths = get_us_sim_and_seg_paths(path, source, split, download)
172
173    if resize_inputs:
174        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
175        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
176            kwargs=kwargs,
177            patch_shape=patch_shape,
178            resize_inputs=resize_inputs,
179            resize_kwargs=resize_kwargs,
180            ensure_rgb=to_rgb,
181        )
182
183    return torch_em.default_segmentation_dataset(
184        raw_paths=image_paths,
185        raw_key=None,
186        label_paths=gt_paths,
187        label_key=None,
188        patch_shape=patch_shape,
189        is_seg_dataset=False,
190        **kwargs
191    )
192
193
194def get_us_sim_and_seg_loader(
195    path: Union[os.PathLike, str],
196    batch_size: int,
197    patch_shape: Tuple[int, int],
198    source: Literal["aus", "rus"],
199    split: Literal["train", "test"],
200    resize_inputs: bool = False,
201    download: bool = False,
202    **kwargs
203) -> DataLoader:
204    """Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images.
205
206    Args:
207        path: Filepath to a folder where the data is downloaded for further processing.
208        batch_size: The batch size for training.
209        patch_shape: The patch shape to use for training.
210        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
211        split: The choice of data split. Either 'train' or 'test'.
212        resize_inputs: Whether to resize the inputs to the patch shape.
213        download: Whether to download the data if it is not present.
214        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
215
216    Returns:
217        The DataLoader.
218    """
219    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
220    dataset = get_us_sim_and_seg_dataset(path, patch_shape, source, split, resize_inputs, download, **ds_kwargs)
221    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
LABEL_MAPS = {(0, 0, 0): 0, (10, 10, 10): 0, (100, 0, 100): 1, (255, 255, 0): 2, (0, 0, 255): 3, (255, 0, 0): 4, (0, 255, 255): 5, (0, 255, 0): 6, (255, 255, 255): 7, (255, 0, 255): 8}
def get_us_sim_and_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48def get_us_sim_and_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
49    """Download the US Simulation & Segmentation dataset.
50
51    Args:
52        path: Filepath to a folder where the data is downloaded for further processing.
53        download: Whether to download the data if it is not present.
54
55    Returns:
56        Filepath where the data is downloaded.
57    """
58    data_dir = os.path.join(path, "abdominal_US", "abdominal_US")
59    if os.path.exists(data_dir):
60        return data_dir
61
62    os.makedirs(path, exist_ok=True)
63
64    zip_path = os.path.join(path, "ussimandsegm.zip")
65    util.download_source_kaggle(path=path, dataset_name="ignaciorlando/ussimandsegm", download=download)
66    util.unzip(zip_path=zip_path, dst=path)
67
68    return data_dir

Download the US Simulation & Segmentation dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_us_sim_and_seg_paths( path: Union[os.PathLike, str], source: Literal['aus', 'rus'], split: Literal['train', 'test'], download: bool = False) -> Tuple[List[str], List[str]]:
100def get_us_sim_and_seg_paths(
101    path: Union[os.PathLike, str],
102    source: Literal["aus", "rus"],
103    split: Literal["train", "test"],
104    download: bool = False,
105) -> Tuple[List[str], List[str]]:
106    """Get paths to the US Simulation & Segmentation data.
107
108    Args:
109        path: Filepath to a folder where the data is downloaded for further processing.
110        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
111        split: The choice of data split. Either 'train' or 'test'.
112        download: Whether to download the data if it is not present.
113
114    Returns:
115        List of filepaths for the image data.
116        List of filepaths for the label data.
117    """
118    if source not in ("aus", "rus"):
119        raise ValueError(f"'{source}' is not a valid source. Choose either 'aus' or 'rus'.")
120
121    if split not in ("train", "test"):
122        raise ValueError(f"'{split}' is not a valid split. Choose either 'train' or 'test'.")
123
124    data_dir = get_us_sim_and_seg_data(path, download)
125
126    image_dir = os.path.join(data_dir, source.upper(), "images", split)
127    gt_dir = os.path.join(data_dir, source.upper(), "annotations", split)
128
129    if not os.path.exists(gt_dir):
130        raise ValueError(f"There are no annotations for the '{source}' source and the '{split}' split.")
131
132    image_paths = natsorted(glob(os.path.join(image_dir, "*")))
133    gt_paths = natsorted(glob(os.path.join(gt_dir, "*")))
134
135    image_stems = {Path(p).stem: p for p in image_paths}
136    gt_stems = {Path(p).stem: p for p in gt_paths}
137    matched_stems = natsorted(set(image_stems) & set(gt_stems))
138    assert len(matched_stems) > 0
139
140    image_paths = [image_stems[stem] for stem in matched_stems]
141    gt_paths = [gt_stems[stem] for stem in matched_stems]
142
143    neu_gt_dir = os.path.join(data_dir, source.upper(), "preprocessed", split)
144    neu_gt_paths = _preprocess_labels(image_paths, gt_paths, neu_gt_dir)
145
146    return image_paths, neu_gt_paths

Get paths to the US Simulation & Segmentation data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
  • split: The choice of data split. Either 'train' or 'test'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_us_sim_and_seg_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], source: Literal['aus', 'rus'], split: Literal['train', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
149def get_us_sim_and_seg_dataset(
150    path: Union[os.PathLike, str],
151    patch_shape: Tuple[int, int],
152    source: Literal["aus", "rus"],
153    split: Literal["train", "test"],
154    resize_inputs: bool = False,
155    download: bool = False,
156    **kwargs
157) -> Dataset:
158    """Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        patch_shape: The patch shape to use for training.
163        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
164        split: The choice of data split. Either 'train' or 'test'.
165        resize_inputs: Whether to resize the inputs to the patch shape.
166        download: Whether to download the data if it is not present.
167        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
168
169    Returns:
170        The segmentation dataset.
171    """
172    image_paths, gt_paths = get_us_sim_and_seg_paths(path, source, split, download)
173
174    if resize_inputs:
175        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
176        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
177            kwargs=kwargs,
178            patch_shape=patch_shape,
179            resize_inputs=resize_inputs,
180            resize_kwargs=resize_kwargs,
181            ensure_rgb=to_rgb,
182        )
183
184    return torch_em.default_segmentation_dataset(
185        raw_paths=image_paths,
186        raw_key=None,
187        label_paths=gt_paths,
188        label_key=None,
189        patch_shape=patch_shape,
190        is_seg_dataset=False,
191        **kwargs
192    )

Get the US Simulation & Segmentation dataset for abdominal organ segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
  • split: The choice of data split. Either 'train' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_us_sim_and_seg_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], source: Literal['aus', 'rus'], split: Literal['train', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
195def get_us_sim_and_seg_loader(
196    path: Union[os.PathLike, str],
197    batch_size: int,
198    patch_shape: Tuple[int, int],
199    source: Literal["aus", "rus"],
200    split: Literal["train", "test"],
201    resize_inputs: bool = False,
202    download: bool = False,
203    **kwargs
204) -> DataLoader:
205    """Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images.
206
207    Args:
208        path: Filepath to a folder where the data is downloaded for further processing.
209        batch_size: The batch size for training.
210        patch_shape: The patch shape to use for training.
211        source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
212        split: The choice of data split. Either 'train' or 'test'.
213        resize_inputs: Whether to resize the inputs to the patch shape.
214        download: Whether to download the data if it is not present.
215        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
216
217    Returns:
218        The DataLoader.
219    """
220    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
221    dataset = get_us_sim_and_seg_dataset(path, patch_shape, source, split, resize_inputs, download, **ds_kwargs)
222    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)

Get the US Simulation & Segmentation dataloader for abdominal organ segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • source: The choice of data source. Either 'aus' (simulated scans) or 'rus' (real scans).
  • split: The choice of data split. Either 'train' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.