torch_em.data.datasets.medical.plaque_us

The Ar-PlaqSegm1 dataset contains annotations for atherosclerotic plaque segmentation in B-mode vascular ultrasound.

The dataset consists of 541 pairs of B-mode ultrasound images (800 x 800 pixels) acquired in Cordoba, Argentina, each paired with a binary segmentation mask of the atherosclerotic plaque. 201 pairs contain no visible plaque (an empty mask) and 340 pairs show one or more plaques. The ground truth was manually delineated by two experienced physicians and released as a monochrome raw image and its corresponding binary mask (foreground: plaque).

The dataset is located at https://data.mendeley.com/datasets/8srkpz52dy/1 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-026-06952-7. Please cite it if you use this dataset for your research.

  1"""The Ar-PlaqSegm1 dataset contains annotations for atherosclerotic plaque segmentation
  2in B-mode vascular ultrasound.
  3
  4The dataset consists of 541 pairs of B-mode ultrasound images (800 x 800 pixels) acquired in
  5Cordoba, Argentina, each paired with a binary segmentation mask of the atherosclerotic plaque.
  6201 pairs contain no visible plaque (an empty mask) and 340 pairs show one or more plaques. The
  7ground truth was manually delineated by two experienced physicians and released as a monochrome
  8raw image and its corresponding binary mask (foreground: plaque).
  9
 10The dataset is located at https://data.mendeley.com/datasets/8srkpz52dy/1 (CC BY 4.0).
 11This dataset is from the publication https://doi.org/10.1038/s41597-026-06952-7.
 12Please cite it if you use this dataset for your research.
 13"""
 14
 15import os
 16import json
 17from glob import glob
 18from tqdm import tqdm
 19from natsort import natsorted
 20from typing import Union, Tuple, List
 21
 22import imageio.v3 as imageio
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31MANIFEST_URL = "https://data.mendeley.com/public-api/datasets/8srkpz52dy?folder_id=&dataset_version=1"
 32
 33N_IMAGES = 541
 34
 35
 36def _get_manifest(path, download):
 37    manifest_path = os.path.join(path, "manifest.json")
 38    if os.path.exists(manifest_path):
 39        with open(manifest_path, "r") as f:
 40            return json.load(f)
 41
 42    if not download:
 43        raise RuntimeError(f"Cannot find the data at '{path}', but download was set to False.")
 44
 45    import requests
 46
 47    response = requests.get(MANIFEST_URL)
 48    response.raise_for_status()
 49    manifest = response.json()["files"]
 50
 51    os.makedirs(path, exist_ok=True)
 52    with open(manifest_path, "w") as f:
 53        json.dump(manifest, f)
 54
 55    return manifest
 56
 57
 58def get_plaque_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 59    """Download the Ar-PlaqSegm1 dataset.
 60
 61    Args:
 62        path: Filepath to a folder where the data is downloaded for further processing.
 63        download: Whether to download the data if it is not present.
 64
 65    Returns:
 66        Filepath to the folder where the images and masks are stored.
 67    """
 68    data_dir = os.path.join(path, "images")
 69    os.makedirs(data_dir, exist_ok=True)
 70
 71    manifest = _get_manifest(path, download)
 72    for entry in tqdm(manifest, desc="Downloading Ar-PlaqSegm1"):
 73        fpath = os.path.join(data_dir, entry["filename"])
 74        content = entry["content_details"]
 75        util.download_source(
 76            path=fpath, url=content["download_url"], download=download, checksum=content["sha256_hash"]
 77        )
 78
 79    return data_dir
 80
 81
 82def _preprocess_labels(label_paths, data_dir):
 83    # Most masks are single-channel, but a subset are stored as an RGB image with the same binary
 84    # mask duplicated across all three channels, which `default_segmentation_dataset` cannot use
 85    # directly, so all masks are normalized to a single-channel (0, 1) label map.
 86    neu_dir = os.path.join(data_dir, "preprocessed_masks")
 87    os.makedirs(neu_dir, exist_ok=True)
 88
 89    neu_label_paths = []
 90    for label_path in label_paths:
 91        neu_path = os.path.join(neu_dir, os.path.basename(label_path))
 92        if not os.path.exists(neu_path):
 93            mask = imageio.imread(label_path)
 94            if mask.ndim == 3:
 95                mask = mask[..., 0]
 96            imageio.imwrite(neu_path, (mask > 0).astype("uint8"))
 97        neu_label_paths.append(neu_path)
 98
 99    return neu_label_paths
100
101
102def get_plaque_us_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
103    """Get paths to the Ar-PlaqSegm1 data.
104
105    Args:
106        path: Filepath to a folder where the data is downloaded for further processing.
107        download: Whether to download the data if it is not present.
108
109    Returns:
110        List of filepaths for the image data.
111        List of filepaths for the label data.
112    """
113    data_dir = get_plaque_us_data(path, download)
114
115    all_paths = natsorted(glob(os.path.join(data_dir, "*.png")))
116    image_paths = [p for p in all_paths if not p.endswith("_labeled.png")]
117    # One image in the original release is named with a stray trailing space before the extension
118    # ("544 .png"), while its mask is not ("544_labeled.png"), so the image id is stripped of
119    # whitespace before deriving the mask filename.
120    label_paths = [
121        os.path.join(os.path.dirname(p), f"{os.path.basename(p)[:-len('.png')].strip()}_labeled.png")
122        for p in image_paths
123    ]
124
125    assert len(image_paths) == N_IMAGES, f"Expected {N_IMAGES} images, found {len(image_paths)}."
126    assert all(os.path.exists(p) for p in label_paths)
127
128    label_paths = _preprocess_labels(label_paths, os.path.dirname(data_dir))
129
130    return image_paths, label_paths
131
132
133def get_plaque_us_dataset(
134    path: Union[os.PathLike, str],
135    patch_shape: Tuple[int, int],
136    resize_inputs: bool = False,
137    download: bool = False,
138    **kwargs
139) -> Dataset:
140    """Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images.
141
142    Args:
143        path: Filepath to a folder where the data is downloaded for further processing.
144        patch_shape: The patch shape to use for training.
145        resize_inputs: Whether to resize the inputs to the expected patch shape.
146        download: Whether to download the data if it is not present.
147        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
148
149    Returns:
150        The segmentation dataset.
151    """
152    image_paths, label_paths = get_plaque_us_paths(path, download)
153
154    if resize_inputs:
155        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
156        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
157            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
158        )
159
160    return torch_em.default_segmentation_dataset(
161        raw_paths=image_paths,
162        raw_key=None,
163        label_paths=label_paths,
164        label_key=None,
165        is_seg_dataset=False,
166        patch_shape=patch_shape,
167        **kwargs
168    )
169
170
171def get_plaque_us_loader(
172    path: Union[os.PathLike, str],
173    batch_size: int,
174    patch_shape: Tuple[int, int],
175    resize_inputs: bool = False,
176    download: bool = False,
177    **kwargs
178) -> DataLoader:
179    """Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images.
180
181    Args:
182        path: Filepath to a folder where the data is downloaded for further processing.
183        batch_size: The batch size for training.
184        patch_shape: The patch shape to use for training.
185        resize_inputs: Whether to resize the inputs to the expected patch shape.
186        download: Whether to download the data if it is not present.
187        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
188
189    Returns:
190        The DataLoader.
191    """
192    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
193    dataset = get_plaque_us_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
194    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
MANIFEST_URL = 'https://data.mendeley.com/public-api/datasets/8srkpz52dy?folder_id=&dataset_version=1'
N_IMAGES = 541
def get_plaque_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
59def get_plaque_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
60    """Download the Ar-PlaqSegm1 dataset.
61
62    Args:
63        path: Filepath to a folder where the data is downloaded for further processing.
64        download: Whether to download the data if it is not present.
65
66    Returns:
67        Filepath to the folder where the images and masks are stored.
68    """
69    data_dir = os.path.join(path, "images")
70    os.makedirs(data_dir, exist_ok=True)
71
72    manifest = _get_manifest(path, download)
73    for entry in tqdm(manifest, desc="Downloading Ar-PlaqSegm1"):
74        fpath = os.path.join(data_dir, entry["filename"])
75        content = entry["content_details"]
76        util.download_source(
77            path=fpath, url=content["download_url"], download=download, checksum=content["sha256_hash"]
78        )
79
80    return data_dir

Download the Ar-PlaqSegm1 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder where the images and masks are stored.

def get_plaque_us_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
103def get_plaque_us_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
104    """Get paths to the Ar-PlaqSegm1 data.
105
106    Args:
107        path: Filepath to a folder where the data is downloaded for further processing.
108        download: Whether to download the data if it is not present.
109
110    Returns:
111        List of filepaths for the image data.
112        List of filepaths for the label data.
113    """
114    data_dir = get_plaque_us_data(path, download)
115
116    all_paths = natsorted(glob(os.path.join(data_dir, "*.png")))
117    image_paths = [p for p in all_paths if not p.endswith("_labeled.png")]
118    # One image in the original release is named with a stray trailing space before the extension
119    # ("544 .png"), while its mask is not ("544_labeled.png"), so the image id is stripped of
120    # whitespace before deriving the mask filename.
121    label_paths = [
122        os.path.join(os.path.dirname(p), f"{os.path.basename(p)[:-len('.png')].strip()}_labeled.png")
123        for p in image_paths
124    ]
125
126    assert len(image_paths) == N_IMAGES, f"Expected {N_IMAGES} images, found {len(image_paths)}."
127    assert all(os.path.exists(p) for p in label_paths)
128
129    label_paths = _preprocess_labels(label_paths, os.path.dirname(data_dir))
130
131    return image_paths, label_paths

Get paths to the Ar-PlaqSegm1 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_plaque_us_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
134def get_plaque_us_dataset(
135    path: Union[os.PathLike, str],
136    patch_shape: Tuple[int, int],
137    resize_inputs: bool = False,
138    download: bool = False,
139    **kwargs
140) -> Dataset:
141    """Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images.
142
143    Args:
144        path: Filepath to a folder where the data is downloaded for further processing.
145        patch_shape: The patch shape to use for training.
146        resize_inputs: Whether to resize the inputs to the expected patch shape.
147        download: Whether to download the data if it is not present.
148        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
149
150    Returns:
151        The segmentation dataset.
152    """
153    image_paths, label_paths = get_plaque_us_paths(path, download)
154
155    if resize_inputs:
156        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
157        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
158            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
159        )
160
161    return torch_em.default_segmentation_dataset(
162        raw_paths=image_paths,
163        raw_key=None,
164        label_paths=label_paths,
165        label_key=None,
166        is_seg_dataset=False,
167        patch_shape=patch_shape,
168        **kwargs
169    )

Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_plaque_us_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
172def get_plaque_us_loader(
173    path: Union[os.PathLike, str],
174    batch_size: int,
175    patch_shape: Tuple[int, int],
176    resize_inputs: bool = False,
177    download: bool = False,
178    **kwargs
179) -> DataLoader:
180    """Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images.
181
182    Args:
183        path: Filepath to a folder where the data is downloaded for further processing.
184        batch_size: The batch size for training.
185        patch_shape: The patch shape to use for training.
186        resize_inputs: Whether to resize the inputs to the expected patch shape.
187        download: Whether to download the data if it is not present.
188        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
189
190    Returns:
191        The DataLoader.
192    """
193    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
194    dataset = get_plaque_us_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
195    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.