torch_em.data.datasets.medical.muscle_us
The MuscleUS dataset contains annotations for cross-sectional muscle segmentation in musculoskeletal ultrasound images.
The dataset consists of 3,917 transverse ultrasound images of the biceps brachii ('BB'), tibialis anterior ('TA') and gastrocnemius medialis ('GM') muscles, acquired on 1,283 subjects (both healthy and with neuromuscular disease), with manual binary masks of the muscle cross-sectional area.
The dataset is located at https://doi.org/10.17632/3jykz7wz8d.1 and is distributed under the CC BY 4.0 license. The dataset is from the publication https://doi.org/10.1016/j.compbiomed.2021.104623. Please cite it if you use this dataset for your research.
1"""The MuscleUS dataset contains annotations for cross-sectional muscle segmentation in 2musculoskeletal ultrasound images. 3 4The dataset consists of 3,917 transverse ultrasound images of the biceps brachii ('BB'), tibialis 5anterior ('TA') and gastrocnemius medialis ('GM') muscles, acquired on 1,283 subjects (both healthy 6and with neuromuscular disease), with manual binary masks of the muscle cross-sectional area. 7 8The dataset is located at https://doi.org/10.17632/3jykz7wz8d.1 and is distributed under the 9CC BY 4.0 license. 10The dataset is from the publication https://doi.org/10.1016/j.compbiomed.2021.104623. 11Please cite it if you use this dataset for your research. 12""" 13 14import os 15from glob import glob 16from natsort import natsorted 17from typing import Union, Tuple, Literal, List, Optional 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24from ..light_microscopy.neurips_cell_seg import to_rgb 25 26 27URL = "https://data.mendeley.com/public-files/datasets/3jykz7wz8d/files/b1601a22-98a5-41ce-9fbb-54dd21834b55/file_downloaded" # noqa 28CHECKSUM = "81dbec46e9f3035831f97aa828735cd6ba04171a72902a9416e7847e75148a11" 29 30MUSCLES = ["BB", "GM", "TA"] 31"""The choice of muscles in the dataset: biceps brachii ('BB'), gastrocnemius medialis ('GM') 32and tibialis anterior ('TA').""" 33 34CATEGORIES = ["Healthy", "Pathological"] 35"""The choice of subject categories in the dataset.""" 36 37 38def get_muscle_us_data(path: Union[os.PathLike, str], download: bool = False) -> str: 39 """Download the MuscleUS dataset. 40 41 Args: 42 path: Filepath to a folder where the data is downloaded for further processing. 43 download: Whether to download the data if it is not present. 44 45 Returns: 46 Filepath where the data is downloaded. 47 """ 48 data_dir = os.path.join(path, "Polito-Radboud-DeepLearningUS") 49 if os.path.exists(data_dir): 50 return data_dir 51 52 os.makedirs(path, exist_ok=True) 53 54 zip_path = os.path.join(path, "muscle_us.zip") 55 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 56 util.unzip(zip_path=zip_path, dst=path) 57 58 return data_dir 59 60 61def get_muscle_us_paths( 62 path: Union[os.PathLike, str], 63 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 64 category: Optional[Literal["Healthy", "Pathological"]] = None, 65 download: bool = False, 66) -> Tuple[List[str], List[str]]: 67 """Get paths to the MuscleUS data. 68 69 Args: 70 path: Filepath to a folder where the data is downloaded for further processing. 71 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 72 category: The choice of subject category. One of 'Healthy', 'Pathological'. 73 By default, loads images of all categories. 74 download: Whether to download the data if it is not present. 75 76 Returns: 77 List of filepaths for the image data. 78 List of filepaths for the label data. 79 """ 80 if muscle is None: 81 muscles = MUSCLES 82 elif muscle in MUSCLES: 83 muscles = [muscle] 84 else: 85 raise ValueError(f"'{muscle}' is not a valid muscle. Choose from {MUSCLES}.") 86 87 if category is None: 88 categories = CATEGORIES 89 elif category in CATEGORIES: 90 categories = [category] 91 else: 92 raise ValueError(f"'{category}' is not a valid category. Choose from {CATEGORIES}.") 93 94 data_dir = get_muscle_us_data(path, download) 95 96 image_paths, label_paths = [], [] 97 for m in muscles: 98 for c in categories: 99 image_dir = os.path.join(data_dir, m, c, "Images") 100 label_dir = os.path.join(data_dir, m, c, "Masks") 101 for image_path in natsorted(glob(os.path.join(image_dir, "*.png"))): 102 label_path = os.path.join(label_dir, os.path.basename(image_path)) 103 if os.path.exists(label_path): 104 image_paths.append(image_path) 105 label_paths.append(label_path) 106 107 return image_paths, label_paths 108 109 110def get_muscle_us_dataset( 111 path: Union[os.PathLike, str], 112 patch_shape: Tuple[int, int], 113 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 114 category: Optional[Literal["Healthy", "Pathological"]] = None, 115 resize_inputs: bool = False, 116 download: bool = False, 117 **kwargs 118) -> Dataset: 119 """Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images. 120 121 Args: 122 path: Filepath to a folder where the data is downloaded for further processing. 123 patch_shape: The patch shape to use for training. 124 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 125 category: The choice of subject category. One of 'Healthy', 'Pathological'. 126 By default, loads images of all categories. 127 resize_inputs: Whether to resize the inputs to the expected patch shape. 128 download: Whether to download the data if it is not present. 129 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 130 131 Returns: 132 The segmentation dataset. 133 """ 134 image_paths, label_paths = get_muscle_us_paths(path, muscle, category, download) 135 136 if resize_inputs: 137 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 138 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 139 kwargs=kwargs, 140 patch_shape=patch_shape, 141 resize_inputs=resize_inputs, 142 resize_kwargs=resize_kwargs, 143 ensure_rgb=to_rgb, 144 ) 145 146 return torch_em.default_segmentation_dataset( 147 raw_paths=image_paths, 148 raw_key=None, 149 label_paths=label_paths, 150 label_key=None, 151 patch_shape=patch_shape, 152 is_seg_dataset=False, 153 **kwargs 154 ) 155 156 157def get_muscle_us_loader( 158 path: Union[os.PathLike, str], 159 batch_size: int, 160 patch_shape: Tuple[int, int], 161 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 162 category: Optional[Literal["Healthy", "Pathological"]] = None, 163 resize_inputs: bool = False, 164 download: bool = False, 165 **kwargs 166) -> DataLoader: 167 """Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images. 168 169 Args: 170 path: Filepath to a folder where the data is downloaded for further processing. 171 batch_size: The batch size for training. 172 patch_shape: The patch shape to use for training. 173 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 174 category: The choice of subject category. One of 'Healthy', 'Pathological'. 175 By default, loads images of all categories. 176 resize_inputs: Whether to resize the inputs to the expected patch shape. 177 download: Whether to download the data if it is not present. 178 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 179 180 Returns: 181 The DataLoader. 182 """ 183 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 184 dataset = get_muscle_us_dataset(path, patch_shape, muscle, category, resize_inputs, download, **ds_kwargs) 185 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
The choice of muscles in the dataset: biceps brachii ('BB'), gastrocnemius medialis ('GM') and tibialis anterior ('TA').
The choice of subject categories in the dataset.
39def get_muscle_us_data(path: Union[os.PathLike, str], download: bool = False) -> str: 40 """Download the MuscleUS dataset. 41 42 Args: 43 path: Filepath to a folder where the data is downloaded for further processing. 44 download: Whether to download the data if it is not present. 45 46 Returns: 47 Filepath where the data is downloaded. 48 """ 49 data_dir = os.path.join(path, "Polito-Radboud-DeepLearningUS") 50 if os.path.exists(data_dir): 51 return data_dir 52 53 os.makedirs(path, exist_ok=True) 54 55 zip_path = os.path.join(path, "muscle_us.zip") 56 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 57 util.unzip(zip_path=zip_path, dst=path) 58 59 return data_dir
Download the MuscleUS dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
62def get_muscle_us_paths( 63 path: Union[os.PathLike, str], 64 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 65 category: Optional[Literal["Healthy", "Pathological"]] = None, 66 download: bool = False, 67) -> Tuple[List[str], List[str]]: 68 """Get paths to the MuscleUS data. 69 70 Args: 71 path: Filepath to a folder where the data is downloaded for further processing. 72 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 73 category: The choice of subject category. One of 'Healthy', 'Pathological'. 74 By default, loads images of all categories. 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 List of filepaths for the image data. 79 List of filepaths for the label data. 80 """ 81 if muscle is None: 82 muscles = MUSCLES 83 elif muscle in MUSCLES: 84 muscles = [muscle] 85 else: 86 raise ValueError(f"'{muscle}' is not a valid muscle. Choose from {MUSCLES}.") 87 88 if category is None: 89 categories = CATEGORIES 90 elif category in CATEGORIES: 91 categories = [category] 92 else: 93 raise ValueError(f"'{category}' is not a valid category. Choose from {CATEGORIES}.") 94 95 data_dir = get_muscle_us_data(path, download) 96 97 image_paths, label_paths = [], [] 98 for m in muscles: 99 for c in categories: 100 image_dir = os.path.join(data_dir, m, c, "Images") 101 label_dir = os.path.join(data_dir, m, c, "Masks") 102 for image_path in natsorted(glob(os.path.join(image_dir, "*.png"))): 103 label_path = os.path.join(label_dir, os.path.basename(image_path)) 104 if os.path.exists(label_path): 105 image_paths.append(image_path) 106 label_paths.append(label_path) 107 108 return image_paths, label_paths
Get paths to the MuscleUS data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
- category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
111def get_muscle_us_dataset( 112 path: Union[os.PathLike, str], 113 patch_shape: Tuple[int, int], 114 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 115 category: Optional[Literal["Healthy", "Pathological"]] = None, 116 resize_inputs: bool = False, 117 download: bool = False, 118 **kwargs 119) -> Dataset: 120 """Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images. 121 122 Args: 123 path: Filepath to a folder where the data is downloaded for further processing. 124 patch_shape: The patch shape to use for training. 125 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 126 category: The choice of subject category. One of 'Healthy', 'Pathological'. 127 By default, loads images of all categories. 128 resize_inputs: Whether to resize the inputs to the expected patch shape. 129 download: Whether to download the data if it is not present. 130 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 131 132 Returns: 133 The segmentation dataset. 134 """ 135 image_paths, label_paths = get_muscle_us_paths(path, muscle, category, download) 136 137 if resize_inputs: 138 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 139 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 140 kwargs=kwargs, 141 patch_shape=patch_shape, 142 resize_inputs=resize_inputs, 143 resize_kwargs=resize_kwargs, 144 ensure_rgb=to_rgb, 145 ) 146 147 return torch_em.default_segmentation_dataset( 148 raw_paths=image_paths, 149 raw_key=None, 150 label_paths=label_paths, 151 label_key=None, 152 patch_shape=patch_shape, 153 is_seg_dataset=False, 154 **kwargs 155 )
Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
- category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
158def get_muscle_us_loader( 159 path: Union[os.PathLike, str], 160 batch_size: int, 161 patch_shape: Tuple[int, int], 162 muscle: Optional[Literal["BB", "GM", "TA"]] = None, 163 category: Optional[Literal["Healthy", "Pathological"]] = None, 164 resize_inputs: bool = False, 165 download: bool = False, 166 **kwargs 167) -> DataLoader: 168 """Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images. 169 170 Args: 171 path: Filepath to a folder where the data is downloaded for further processing. 172 batch_size: The batch size for training. 173 patch_shape: The patch shape to use for training. 174 muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles. 175 category: The choice of subject category. One of 'Healthy', 'Pathological'. 176 By default, loads images of all categories. 177 resize_inputs: Whether to resize the inputs to the expected patch shape. 178 download: Whether to download the data if it is not present. 179 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 180 181 Returns: 182 The DataLoader. 183 """ 184 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 185 dataset = get_muscle_us_dataset(path, patch_shape, muscle, category, resize_inputs, download, **ds_kwargs) 186 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
- category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.