torch_em.data.datasets.medical.plaque_us
The Ar-PlaqSegm1 dataset contains annotations for atherosclerotic plaque segmentation in B-mode vascular ultrasound.
The dataset consists of 541 pairs of B-mode ultrasound images (800 x 800 pixels) acquired in Cordoba, Argentina, each paired with a binary segmentation mask of the atherosclerotic plaque. 201 pairs contain no visible plaque (an empty mask) and 340 pairs show one or more plaques. The ground truth was manually delineated by two experienced physicians and released as a monochrome raw image and its corresponding binary mask (foreground: plaque).
The dataset is located at https://data.mendeley.com/datasets/8srkpz52dy/1 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-026-06952-7. Please cite it if you use this dataset for your research.
1"""The Ar-PlaqSegm1 dataset contains annotations for atherosclerotic plaque segmentation 2in B-mode vascular ultrasound. 3 4The dataset consists of 541 pairs of B-mode ultrasound images (800 x 800 pixels) acquired in 5Cordoba, Argentina, each paired with a binary segmentation mask of the atherosclerotic plaque. 6201 pairs contain no visible plaque (an empty mask) and 340 pairs show one or more plaques. The 7ground truth was manually delineated by two experienced physicians and released as a monochrome 8raw image and its corresponding binary mask (foreground: plaque). 9 10The dataset is located at https://data.mendeley.com/datasets/8srkpz52dy/1 (CC BY 4.0). 11This dataset is from the publication https://doi.org/10.1038/s41597-026-06952-7. 12Please cite it if you use this dataset for your research. 13""" 14 15import os 16import json 17from glob import glob 18from tqdm import tqdm 19from natsort import natsorted 20from typing import Union, Tuple, List 21 22import imageio.v3 as imageio 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31MANIFEST_URL = "https://data.mendeley.com/public-api/datasets/8srkpz52dy?folder_id=&dataset_version=1" 32 33N_IMAGES = 541 34 35 36def _get_manifest(path, download): 37 manifest_path = os.path.join(path, "manifest.json") 38 if os.path.exists(manifest_path): 39 with open(manifest_path, "r") as f: 40 return json.load(f) 41 42 if not download: 43 raise RuntimeError(f"Cannot find the data at '{path}', but download was set to False.") 44 45 import requests 46 47 response = requests.get(MANIFEST_URL) 48 response.raise_for_status() 49 manifest = response.json()["files"] 50 51 os.makedirs(path, exist_ok=True) 52 with open(manifest_path, "w") as f: 53 json.dump(manifest, f) 54 55 return manifest 56 57 58def get_plaque_us_data(path: Union[os.PathLike, str], download: bool = False) -> str: 59 """Download the Ar-PlaqSegm1 dataset. 60 61 Args: 62 path: Filepath to a folder where the data is downloaded for further processing. 63 download: Whether to download the data if it is not present. 64 65 Returns: 66 Filepath to the folder where the images and masks are stored. 67 """ 68 data_dir = os.path.join(path, "images") 69 os.makedirs(data_dir, exist_ok=True) 70 71 manifest = _get_manifest(path, download) 72 for entry in tqdm(manifest, desc="Downloading Ar-PlaqSegm1"): 73 fpath = os.path.join(data_dir, entry["filename"]) 74 content = entry["content_details"] 75 util.download_source( 76 path=fpath, url=content["download_url"], download=download, checksum=content["sha256_hash"] 77 ) 78 79 return data_dir 80 81 82def _preprocess_labels(label_paths, data_dir): 83 # Most masks are single-channel, but a subset are stored as an RGB image with the same binary 84 # mask duplicated across all three channels, which `default_segmentation_dataset` cannot use 85 # directly, so all masks are normalized to a single-channel (0, 1) label map. 86 neu_dir = os.path.join(data_dir, "preprocessed_masks") 87 os.makedirs(neu_dir, exist_ok=True) 88 89 neu_label_paths = [] 90 for label_path in label_paths: 91 neu_path = os.path.join(neu_dir, os.path.basename(label_path)) 92 if not os.path.exists(neu_path): 93 mask = imageio.imread(label_path) 94 if mask.ndim == 3: 95 mask = mask[..., 0] 96 imageio.imwrite(neu_path, (mask > 0).astype("uint8")) 97 neu_label_paths.append(neu_path) 98 99 return neu_label_paths 100 101 102def get_plaque_us_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 103 """Get paths to the Ar-PlaqSegm1 data. 104 105 Args: 106 path: Filepath to a folder where the data is downloaded for further processing. 107 download: Whether to download the data if it is not present. 108 109 Returns: 110 List of filepaths for the image data. 111 List of filepaths for the label data. 112 """ 113 data_dir = get_plaque_us_data(path, download) 114 115 all_paths = natsorted(glob(os.path.join(data_dir, "*.png"))) 116 image_paths = [p for p in all_paths if not p.endswith("_labeled.png")] 117 # One image in the original release is named with a stray trailing space before the extension 118 # ("544 .png"), while its mask is not ("544_labeled.png"), so the image id is stripped of 119 # whitespace before deriving the mask filename. 120 label_paths = [ 121 os.path.join(os.path.dirname(p), f"{os.path.basename(p)[:-len('.png')].strip()}_labeled.png") 122 for p in image_paths 123 ] 124 125 assert len(image_paths) == N_IMAGES, f"Expected {N_IMAGES} images, found {len(image_paths)}." 126 assert all(os.path.exists(p) for p in label_paths) 127 128 label_paths = _preprocess_labels(label_paths, os.path.dirname(data_dir)) 129 130 return image_paths, label_paths 131 132 133def get_plaque_us_dataset( 134 path: Union[os.PathLike, str], 135 patch_shape: Tuple[int, int], 136 resize_inputs: bool = False, 137 download: bool = False, 138 **kwargs 139) -> Dataset: 140 """Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images. 141 142 Args: 143 path: Filepath to a folder where the data is downloaded for further processing. 144 patch_shape: The patch shape to use for training. 145 resize_inputs: Whether to resize the inputs to the expected patch shape. 146 download: Whether to download the data if it is not present. 147 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 148 149 Returns: 150 The segmentation dataset. 151 """ 152 image_paths, label_paths = get_plaque_us_paths(path, download) 153 154 if resize_inputs: 155 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 156 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 157 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 158 ) 159 160 return torch_em.default_segmentation_dataset( 161 raw_paths=image_paths, 162 raw_key=None, 163 label_paths=label_paths, 164 label_key=None, 165 is_seg_dataset=False, 166 patch_shape=patch_shape, 167 **kwargs 168 ) 169 170 171def get_plaque_us_loader( 172 path: Union[os.PathLike, str], 173 batch_size: int, 174 patch_shape: Tuple[int, int], 175 resize_inputs: bool = False, 176 download: bool = False, 177 **kwargs 178) -> DataLoader: 179 """Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images. 180 181 Args: 182 path: Filepath to a folder where the data is downloaded for further processing. 183 batch_size: The batch size for training. 184 patch_shape: The patch shape to use for training. 185 resize_inputs: Whether to resize the inputs to the expected patch shape. 186 download: Whether to download the data if it is not present. 187 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 188 189 Returns: 190 The DataLoader. 191 """ 192 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 193 dataset = get_plaque_us_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 194 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
59def get_plaque_us_data(path: Union[os.PathLike, str], download: bool = False) -> str: 60 """Download the Ar-PlaqSegm1 dataset. 61 62 Args: 63 path: Filepath to a folder where the data is downloaded for further processing. 64 download: Whether to download the data if it is not present. 65 66 Returns: 67 Filepath to the folder where the images and masks are stored. 68 """ 69 data_dir = os.path.join(path, "images") 70 os.makedirs(data_dir, exist_ok=True) 71 72 manifest = _get_manifest(path, download) 73 for entry in tqdm(manifest, desc="Downloading Ar-PlaqSegm1"): 74 fpath = os.path.join(data_dir, entry["filename"]) 75 content = entry["content_details"] 76 util.download_source( 77 path=fpath, url=content["download_url"], download=download, checksum=content["sha256_hash"] 78 ) 79 80 return data_dir
Download the Ar-PlaqSegm1 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder where the images and masks are stored.
103def get_plaque_us_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 104 """Get paths to the Ar-PlaqSegm1 data. 105 106 Args: 107 path: Filepath to a folder where the data is downloaded for further processing. 108 download: Whether to download the data if it is not present. 109 110 Returns: 111 List of filepaths for the image data. 112 List of filepaths for the label data. 113 """ 114 data_dir = get_plaque_us_data(path, download) 115 116 all_paths = natsorted(glob(os.path.join(data_dir, "*.png"))) 117 image_paths = [p for p in all_paths if not p.endswith("_labeled.png")] 118 # One image in the original release is named with a stray trailing space before the extension 119 # ("544 .png"), while its mask is not ("544_labeled.png"), so the image id is stripped of 120 # whitespace before deriving the mask filename. 121 label_paths = [ 122 os.path.join(os.path.dirname(p), f"{os.path.basename(p)[:-len('.png')].strip()}_labeled.png") 123 for p in image_paths 124 ] 125 126 assert len(image_paths) == N_IMAGES, f"Expected {N_IMAGES} images, found {len(image_paths)}." 127 assert all(os.path.exists(p) for p in label_paths) 128 129 label_paths = _preprocess_labels(label_paths, os.path.dirname(data_dir)) 130 131 return image_paths, label_paths
Get paths to the Ar-PlaqSegm1 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
134def get_plaque_us_dataset( 135 path: Union[os.PathLike, str], 136 patch_shape: Tuple[int, int], 137 resize_inputs: bool = False, 138 download: bool = False, 139 **kwargs 140) -> Dataset: 141 """Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images. 142 143 Args: 144 path: Filepath to a folder where the data is downloaded for further processing. 145 patch_shape: The patch shape to use for training. 146 resize_inputs: Whether to resize the inputs to the expected patch shape. 147 download: Whether to download the data if it is not present. 148 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 149 150 Returns: 151 The segmentation dataset. 152 """ 153 image_paths, label_paths = get_plaque_us_paths(path, download) 154 155 if resize_inputs: 156 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 157 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 158 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 159 ) 160 161 return torch_em.default_segmentation_dataset( 162 raw_paths=image_paths, 163 raw_key=None, 164 label_paths=label_paths, 165 label_key=None, 166 is_seg_dataset=False, 167 patch_shape=patch_shape, 168 **kwargs 169 )
Get the Ar-PlaqSegm1 dataset for atherosclerotic plaque segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
172def get_plaque_us_loader( 173 path: Union[os.PathLike, str], 174 batch_size: int, 175 patch_shape: Tuple[int, int], 176 resize_inputs: bool = False, 177 download: bool = False, 178 **kwargs 179) -> DataLoader: 180 """Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images. 181 182 Args: 183 path: Filepath to a folder where the data is downloaded for further processing. 184 batch_size: The batch size for training. 185 patch_shape: The patch shape to use for training. 186 resize_inputs: Whether to resize the inputs to the expected patch shape. 187 download: Whether to download the data if it is not present. 188 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 189 190 Returns: 191 The DataLoader. 192 """ 193 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 194 dataset = get_plaque_us_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 195 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Ar-PlaqSegm1 dataloader for atherosclerotic plaque segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.