torch_em.data.datasets.medical.muscle_us

The MuscleUS dataset contains annotations for cross-sectional muscle segmentation in musculoskeletal ultrasound images.

The dataset consists of 3,917 transverse ultrasound images of the biceps brachii ('BB'), tibialis anterior ('TA') and gastrocnemius medialis ('GM') muscles, acquired on 1,283 subjects (both healthy and with neuromuscular disease), with manual binary masks of the muscle cross-sectional area.

The dataset is located at https://doi.org/10.17632/3jykz7wz8d.1 and is distributed under the CC BY 4.0 license. The dataset is from the publication https://doi.org/10.1016/j.compbiomed.2021.104623. Please cite it if you use this dataset for your research.

  1"""The MuscleUS dataset contains annotations for cross-sectional muscle segmentation in
  2musculoskeletal ultrasound images.
  3
  4The dataset consists of 3,917 transverse ultrasound images of the biceps brachii ('BB'), tibialis
  5anterior ('TA') and gastrocnemius medialis ('GM') muscles, acquired on 1,283 subjects (both healthy
  6and with neuromuscular disease), with manual binary masks of the muscle cross-sectional area.
  7
  8The dataset is located at https://doi.org/10.17632/3jykz7wz8d.1 and is distributed under the
  9CC BY 4.0 license.
 10The dataset is from the publication https://doi.org/10.1016/j.compbiomed.2021.104623.
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15from glob import glob
 16from natsort import natsorted
 17from typing import Union, Tuple, Literal, List, Optional
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24from ..light_microscopy.neurips_cell_seg import to_rgb
 25
 26
 27URL = "https://data.mendeley.com/public-files/datasets/3jykz7wz8d/files/b1601a22-98a5-41ce-9fbb-54dd21834b55/file_downloaded"  # noqa
 28CHECKSUM = "81dbec46e9f3035831f97aa828735cd6ba04171a72902a9416e7847e75148a11"
 29
 30MUSCLES = ["BB", "GM", "TA"]
 31"""The choice of muscles in the dataset: biceps brachii ('BB'), gastrocnemius medialis ('GM')
 32and tibialis anterior ('TA')."""
 33
 34CATEGORIES = ["Healthy", "Pathological"]
 35"""The choice of subject categories in the dataset."""
 36
 37
 38def get_muscle_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 39    """Download the MuscleUS dataset.
 40
 41    Args:
 42        path: Filepath to a folder where the data is downloaded for further processing.
 43        download: Whether to download the data if it is not present.
 44
 45    Returns:
 46        Filepath where the data is downloaded.
 47    """
 48    data_dir = os.path.join(path, "Polito-Radboud-DeepLearningUS")
 49    if os.path.exists(data_dir):
 50        return data_dir
 51
 52    os.makedirs(path, exist_ok=True)
 53
 54    zip_path = os.path.join(path, "muscle_us.zip")
 55    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 56    util.unzip(zip_path=zip_path, dst=path)
 57
 58    return data_dir
 59
 60
 61def get_muscle_us_paths(
 62    path: Union[os.PathLike, str],
 63    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
 64    category: Optional[Literal["Healthy", "Pathological"]] = None,
 65    download: bool = False,
 66) -> Tuple[List[str], List[str]]:
 67    """Get paths to the MuscleUS data.
 68
 69    Args:
 70        path: Filepath to a folder where the data is downloaded for further processing.
 71        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
 72        category: The choice of subject category. One of 'Healthy', 'Pathological'.
 73            By default, loads images of all categories.
 74        download: Whether to download the data if it is not present.
 75
 76    Returns:
 77        List of filepaths for the image data.
 78        List of filepaths for the label data.
 79    """
 80    if muscle is None:
 81        muscles = MUSCLES
 82    elif muscle in MUSCLES:
 83        muscles = [muscle]
 84    else:
 85        raise ValueError(f"'{muscle}' is not a valid muscle. Choose from {MUSCLES}.")
 86
 87    if category is None:
 88        categories = CATEGORIES
 89    elif category in CATEGORIES:
 90        categories = [category]
 91    else:
 92        raise ValueError(f"'{category}' is not a valid category. Choose from {CATEGORIES}.")
 93
 94    data_dir = get_muscle_us_data(path, download)
 95
 96    image_paths, label_paths = [], []
 97    for m in muscles:
 98        for c in categories:
 99            image_dir = os.path.join(data_dir, m, c, "Images")
100            label_dir = os.path.join(data_dir, m, c, "Masks")
101            for image_path in natsorted(glob(os.path.join(image_dir, "*.png"))):
102                label_path = os.path.join(label_dir, os.path.basename(image_path))
103                if os.path.exists(label_path):
104                    image_paths.append(image_path)
105                    label_paths.append(label_path)
106
107    return image_paths, label_paths
108
109
110def get_muscle_us_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
114    category: Optional[Literal["Healthy", "Pathological"]] = None,
115    resize_inputs: bool = False,
116    download: bool = False,
117    **kwargs
118) -> Dataset:
119    """Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images.
120
121    Args:
122        path: Filepath to a folder where the data is downloaded for further processing.
123        patch_shape: The patch shape to use for training.
124        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
125        category: The choice of subject category. One of 'Healthy', 'Pathological'.
126            By default, loads images of all categories.
127        resize_inputs: Whether to resize the inputs to the expected patch shape.
128        download: Whether to download the data if it is not present.
129        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
130
131    Returns:
132        The segmentation dataset.
133    """
134    image_paths, label_paths = get_muscle_us_paths(path, muscle, category, download)
135
136    if resize_inputs:
137        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
138        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
139            kwargs=kwargs,
140            patch_shape=patch_shape,
141            resize_inputs=resize_inputs,
142            resize_kwargs=resize_kwargs,
143            ensure_rgb=to_rgb,
144        )
145
146    return torch_em.default_segmentation_dataset(
147        raw_paths=image_paths,
148        raw_key=None,
149        label_paths=label_paths,
150        label_key=None,
151        patch_shape=patch_shape,
152        is_seg_dataset=False,
153        **kwargs
154    )
155
156
157def get_muscle_us_loader(
158    path: Union[os.PathLike, str],
159    batch_size: int,
160    patch_shape: Tuple[int, int],
161    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
162    category: Optional[Literal["Healthy", "Pathological"]] = None,
163    resize_inputs: bool = False,
164    download: bool = False,
165    **kwargs
166) -> DataLoader:
167    """Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images.
168
169    Args:
170        path: Filepath to a folder where the data is downloaded for further processing.
171        batch_size: The batch size for training.
172        patch_shape: The patch shape to use for training.
173        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
174        category: The choice of subject category. One of 'Healthy', 'Pathological'.
175            By default, loads images of all categories.
176        resize_inputs: Whether to resize the inputs to the expected patch shape.
177        download: Whether to download the data if it is not present.
178        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
179
180    Returns:
181        The DataLoader.
182    """
183    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
184    dataset = get_muscle_us_dataset(path, patch_shape, muscle, category, resize_inputs, download, **ds_kwargs)
185    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://data.mendeley.com/public-files/datasets/3jykz7wz8d/files/b1601a22-98a5-41ce-9fbb-54dd21834b55/file_downloaded'
CHECKSUM = '81dbec46e9f3035831f97aa828735cd6ba04171a72902a9416e7847e75148a11'
MUSCLES = ['BB', 'GM', 'TA']

The choice of muscles in the dataset: biceps brachii ('BB'), gastrocnemius medialis ('GM') and tibialis anterior ('TA').

CATEGORIES = ['Healthy', 'Pathological']

The choice of subject categories in the dataset.

def get_muscle_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39def get_muscle_us_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40    """Download the MuscleUS dataset.
41
42    Args:
43        path: Filepath to a folder where the data is downloaded for further processing.
44        download: Whether to download the data if it is not present.
45
46    Returns:
47        Filepath where the data is downloaded.
48    """
49    data_dir = os.path.join(path, "Polito-Radboud-DeepLearningUS")
50    if os.path.exists(data_dir):
51        return data_dir
52
53    os.makedirs(path, exist_ok=True)
54
55    zip_path = os.path.join(path, "muscle_us.zip")
56    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
57    util.unzip(zip_path=zip_path, dst=path)
58
59    return data_dir

Download the MuscleUS dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_muscle_us_paths( path: Union[os.PathLike, str], muscle: Optional[Literal['BB', 'GM', 'TA']] = None, category: Optional[Literal['Healthy', 'Pathological']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 62def get_muscle_us_paths(
 63    path: Union[os.PathLike, str],
 64    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
 65    category: Optional[Literal["Healthy", "Pathological"]] = None,
 66    download: bool = False,
 67) -> Tuple[List[str], List[str]]:
 68    """Get paths to the MuscleUS data.
 69
 70    Args:
 71        path: Filepath to a folder where the data is downloaded for further processing.
 72        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
 73        category: The choice of subject category. One of 'Healthy', 'Pathological'.
 74            By default, loads images of all categories.
 75        download: Whether to download the data if it is not present.
 76
 77    Returns:
 78        List of filepaths for the image data.
 79        List of filepaths for the label data.
 80    """
 81    if muscle is None:
 82        muscles = MUSCLES
 83    elif muscle in MUSCLES:
 84        muscles = [muscle]
 85    else:
 86        raise ValueError(f"'{muscle}' is not a valid muscle. Choose from {MUSCLES}.")
 87
 88    if category is None:
 89        categories = CATEGORIES
 90    elif category in CATEGORIES:
 91        categories = [category]
 92    else:
 93        raise ValueError(f"'{category}' is not a valid category. Choose from {CATEGORIES}.")
 94
 95    data_dir = get_muscle_us_data(path, download)
 96
 97    image_paths, label_paths = [], []
 98    for m in muscles:
 99        for c in categories:
100            image_dir = os.path.join(data_dir, m, c, "Images")
101            label_dir = os.path.join(data_dir, m, c, "Masks")
102            for image_path in natsorted(glob(os.path.join(image_dir, "*.png"))):
103                label_path = os.path.join(label_dir, os.path.basename(image_path))
104                if os.path.exists(label_path):
105                    image_paths.append(image_path)
106                    label_paths.append(label_path)
107
108    return image_paths, label_paths

Get paths to the MuscleUS data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
  • category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_muscle_us_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], muscle: Optional[Literal['BB', 'GM', 'TA']] = None, category: Optional[Literal['Healthy', 'Pathological']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
111def get_muscle_us_dataset(
112    path: Union[os.PathLike, str],
113    patch_shape: Tuple[int, int],
114    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
115    category: Optional[Literal["Healthy", "Pathological"]] = None,
116    resize_inputs: bool = False,
117    download: bool = False,
118    **kwargs
119) -> Dataset:
120    """Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images.
121
122    Args:
123        path: Filepath to a folder where the data is downloaded for further processing.
124        patch_shape: The patch shape to use for training.
125        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
126        category: The choice of subject category. One of 'Healthy', 'Pathological'.
127            By default, loads images of all categories.
128        resize_inputs: Whether to resize the inputs to the expected patch shape.
129        download: Whether to download the data if it is not present.
130        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
131
132    Returns:
133        The segmentation dataset.
134    """
135    image_paths, label_paths = get_muscle_us_paths(path, muscle, category, download)
136
137    if resize_inputs:
138        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
139        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
140            kwargs=kwargs,
141            patch_shape=patch_shape,
142            resize_inputs=resize_inputs,
143            resize_kwargs=resize_kwargs,
144            ensure_rgb=to_rgb,
145        )
146
147    return torch_em.default_segmentation_dataset(
148        raw_paths=image_paths,
149        raw_key=None,
150        label_paths=label_paths,
151        label_key=None,
152        patch_shape=patch_shape,
153        is_seg_dataset=False,
154        **kwargs
155    )

Get the MuscleUS dataset for muscle cross-sectional area segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
  • category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_muscle_us_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], muscle: Optional[Literal['BB', 'GM', 'TA']] = None, category: Optional[Literal['Healthy', 'Pathological']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
158def get_muscle_us_loader(
159    path: Union[os.PathLike, str],
160    batch_size: int,
161    patch_shape: Tuple[int, int],
162    muscle: Optional[Literal["BB", "GM", "TA"]] = None,
163    category: Optional[Literal["Healthy", "Pathological"]] = None,
164    resize_inputs: bool = False,
165    download: bool = False,
166    **kwargs
167) -> DataLoader:
168    """Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images.
169
170    Args:
171        path: Filepath to a folder where the data is downloaded for further processing.
172        batch_size: The batch size for training.
173        patch_shape: The patch shape to use for training.
174        muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
175        category: The choice of subject category. One of 'Healthy', 'Pathological'.
176            By default, loads images of all categories.
177        resize_inputs: Whether to resize the inputs to the expected patch shape.
178        download: Whether to download the data if it is not present.
179        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
180
181    Returns:
182        The DataLoader.
183    """
184    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
185    dataset = get_muscle_us_dataset(path, patch_shape, muscle, category, resize_inputs, download, **ds_kwargs)
186    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the MuscleUS dataloader for muscle cross-sectional area segmentation in ultrasound images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • muscle: The choice of muscle. One of 'BB', 'GM', 'TA'. By default, loads images of all muscles.
  • category: The choice of subject category. One of 'Healthy', 'Pathological'. By default, loads images of all categories.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.