torch_em.data.datasets.medical.bonedat
BoneDat is a dataset for the segmentation of the pelvis, sacrum and L4/L5 vertebrae in lumbopelvic CT.
The dataset consists of 278 anonymized human lumbopelvic CT scans (ages 16-91, balanced by sex), together with multi-label segmentation masks that delineate the individual pelvic bones, the sacrum and the L4/L5 vertebrae. The masks were generated with the Biomedisa algorithm and refined and validated by experts.
The dataset is hosted on Zenodo as 8 linked records. The main record (https://doi.org/10.5281/zenodo.15189761) only holds supplementary results and metadata from the associated publication, not usable image data. The raw CT scans and segmentation masks are distributed across 7 records that together form a single split rar archive (BoneDat.part1.rar - BoneDat.part7.rar, https://doi.org/10.5281/zenodo.15188359 and follow-up DOIs). Most of this archive (parts 1-6, about 290 GB) holds registration and template outputs (deformation fields, warped volumes, meshes) that are not exposed by this module. The raw CT scans and segmentation masks used here are both fully contained within part 7 alone (about 27 GB), so this module only downloads that part.
All 8 records are licensed under CC-BY-4.0. Note that the article text itself is under a separate, more restrictive Springer Nature license, but this does not apply to the dataset files, which are CC-BY-4.0.
This dataset is from the publication https://doi.org/10.1038/s41597-025-05161-y. Please cite it if you use this dataset for your research.
1"""BoneDat is a dataset for the segmentation of the pelvis, sacrum and L4/L5 vertebrae in lumbopelvic CT. 2 3The dataset consists of 278 anonymized human lumbopelvic CT scans (ages 16-91, balanced by sex), together with 4multi-label segmentation masks that delineate the individual pelvic bones, the sacrum and the L4/L5 vertebrae. 5The masks were generated with the Biomedisa algorithm and refined and validated by experts. 6 7The dataset is hosted on Zenodo as 8 linked records. The main record (https://doi.org/10.5281/zenodo.15189761) 8only holds supplementary results and metadata from the associated publication, not usable image data. The raw 9CT scans and segmentation masks are distributed across 7 records that together form a single split rar archive 10(BoneDat.part1.rar - BoneDat.part7.rar, https://doi.org/10.5281/zenodo.15188359 and follow-up DOIs). Most of 11this archive (parts 1-6, about 290 GB) holds registration and template outputs (deformation fields, warped 12volumes, meshes) that are not exposed by this module. The raw CT scans and segmentation masks used here are 13both fully contained within part 7 alone (about 27 GB), so this module only downloads that part. 14 15All 8 records are licensed under CC-BY-4.0. Note that the article text itself is under a separate, more 16restrictive Springer Nature license, but this does not apply to the dataset files, which are CC-BY-4.0. 17 18This dataset is from the publication https://doi.org/10.1038/s41597-025-05161-y. Please cite it if you use this 19dataset for your research. 20""" 21 22import os 23import shutil 24import subprocess 25from glob import glob 26from natsort import natsorted 27from typing import Union, Tuple, List 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36URL = "https://zenodo.org/records/15189605/files/BoneDat.part7.rar?download=1" 37CHECKSUM = "f27bf0813df07c1cceff822246e2793bb03ca74e10f6e7e87861e19b9bdafb85" 38 39 40def _extract_bonedat(rar_path, data_dir): 41 if shutil.which("7z") is None: 42 raise RuntimeError( 43 "Need the 'p7zip' CLI to extract this archive. You can install it via 'conda install -c conda-forge p7zip'." # noqa 44 ) 45 46 # The rar archive is part of a 7-part split archive (this is part 7). Extracting it in isolation (without 47 # the preceding 6 parts) makes '7z' report a non-fatal 'Headers Error' for the missing volume chain, but the 48 # files that are fully contained within this part (i.e. everything under 'raw/' and 'derived/segmentation/') 49 # are still extracted correctly. We therefore do not check the subprocess return code here and instead 50 # verify the presence of the expected output further below. 51 subprocess.run(["7z", "x", f"-o{data_dir}", "-y", rar_path, "raw/*", "derived/segmentation/*"]) 52 53 54def get_bonedat_data(path: Union[os.PathLike, str], download: bool = False) -> str: 55 """Download the BoneDat dataset. 56 57 Args: 58 path: Filepath to a folder where the data is downloaded for further processing. 59 download: Whether to download the data if it is not present. 60 61 Returns: 62 Filepath where the data is downloaded. 63 """ 64 data_dir = os.path.join(path, "data") 65 if os.path.exists(os.path.join(data_dir, "raw")) and os.path.exists(os.path.join(data_dir, "derived")): 66 return data_dir 67 68 os.makedirs(path, exist_ok=True) 69 70 rar_path = os.path.join(path, "BoneDat.part7.rar") 71 util.download_source(path=rar_path, url=URL, download=download, checksum=CHECKSUM) 72 73 os.makedirs(data_dir, exist_ok=True) 74 _extract_bonedat(rar_path, data_dir) 75 76 if not os.path.exists(os.path.join(data_dir, "raw")) or not os.path.exists(os.path.join(data_dir, "derived")): 77 raise RuntimeError(f"Extraction seems to have failed: could not find the expected data at '{data_dir}'.") 78 79 os.remove(rar_path) 80 81 return data_dir 82 83 84def get_bonedat_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 85 """Get paths to the BoneDat data. 86 87 Args: 88 path: Filepath to a folder where the data is downloaded for further processing. 89 download: Whether to download the data if it is not present. 90 91 Returns: 92 List of filepaths for the image data. 93 List of filepaths for the label data. 94 """ 95 data_dir = get_bonedat_data(path, download) 96 97 raw_paths = natsorted(glob(os.path.join(data_dir, "raw", "*", "original.nii.gz"))) 98 label_paths = natsorted(glob(os.path.join(data_dir, "derived", "segmentation", "*", "mask.nii.gz"))) 99 100 if len(raw_paths) == 0 or len(raw_paths) != len(label_paths): 101 raise RuntimeError("Something went wrong with fetching the image and label paths.") 102 103 return raw_paths, label_paths 104 105 106def get_bonedat_dataset( 107 path: Union[os.PathLike, str], 108 patch_shape: Tuple[int, ...], 109 resize_inputs: bool = False, 110 download: bool = False, 111 **kwargs 112) -> Dataset: 113 """Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation. 114 115 Args: 116 path: Filepath to a folder where the data is downloaded for further processing. 117 patch_shape: The patch shape to use for training. 118 resize_inputs: Whether to resize inputs to the desired patch shape. 119 download: Whether to download the data if it is not present. 120 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 121 122 Returns: 123 The segmentation dataset. 124 """ 125 raw_paths, label_paths = get_bonedat_paths(path, download) 126 127 if resize_inputs: 128 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 129 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 130 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 131 ) 132 133 return torch_em.default_segmentation_dataset( 134 raw_paths=raw_paths, 135 raw_key="data", 136 label_paths=label_paths, 137 label_key="data", 138 patch_shape=patch_shape, 139 is_seg_dataset=True, 140 **kwargs 141 ) 142 143 144def get_bonedat_loader( 145 path: Union[os.PathLike, str], 146 batch_size: int, 147 patch_shape: Tuple[int, ...], 148 resize_inputs: bool = False, 149 download: bool = False, 150 **kwargs 151) -> DataLoader: 152 """Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation. 153 154 Args: 155 path: Filepath to a folder where the data is downloaded for further processing. 156 batch_size: The batch size for training. 157 patch_shape: The patch shape to use for training. 158 resize_inputs: Whether to resize inputs to the desired patch shape. 159 download: Whether to download the data if it is not present. 160 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 161 162 Returns: 163 The DataLoader. 164 """ 165 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 166 dataset = get_bonedat_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 167 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
55def get_bonedat_data(path: Union[os.PathLike, str], download: bool = False) -> str: 56 """Download the BoneDat dataset. 57 58 Args: 59 path: Filepath to a folder where the data is downloaded for further processing. 60 download: Whether to download the data if it is not present. 61 62 Returns: 63 Filepath where the data is downloaded. 64 """ 65 data_dir = os.path.join(path, "data") 66 if os.path.exists(os.path.join(data_dir, "raw")) and os.path.exists(os.path.join(data_dir, "derived")): 67 return data_dir 68 69 os.makedirs(path, exist_ok=True) 70 71 rar_path = os.path.join(path, "BoneDat.part7.rar") 72 util.download_source(path=rar_path, url=URL, download=download, checksum=CHECKSUM) 73 74 os.makedirs(data_dir, exist_ok=True) 75 _extract_bonedat(rar_path, data_dir) 76 77 if not os.path.exists(os.path.join(data_dir, "raw")) or not os.path.exists(os.path.join(data_dir, "derived")): 78 raise RuntimeError(f"Extraction seems to have failed: could not find the expected data at '{data_dir}'.") 79 80 os.remove(rar_path) 81 82 return data_dir
Download the BoneDat dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
85def get_bonedat_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 86 """Get paths to the BoneDat data. 87 88 Args: 89 path: Filepath to a folder where the data is downloaded for further processing. 90 download: Whether to download the data if it is not present. 91 92 Returns: 93 List of filepaths for the image data. 94 List of filepaths for the label data. 95 """ 96 data_dir = get_bonedat_data(path, download) 97 98 raw_paths = natsorted(glob(os.path.join(data_dir, "raw", "*", "original.nii.gz"))) 99 label_paths = natsorted(glob(os.path.join(data_dir, "derived", "segmentation", "*", "mask.nii.gz"))) 100 101 if len(raw_paths) == 0 or len(raw_paths) != len(label_paths): 102 raise RuntimeError("Something went wrong with fetching the image and label paths.") 103 104 return raw_paths, label_paths
Get paths to the BoneDat data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
107def get_bonedat_dataset( 108 path: Union[os.PathLike, str], 109 patch_shape: Tuple[int, ...], 110 resize_inputs: bool = False, 111 download: bool = False, 112 **kwargs 113) -> Dataset: 114 """Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation. 115 116 Args: 117 path: Filepath to a folder where the data is downloaded for further processing. 118 patch_shape: The patch shape to use for training. 119 resize_inputs: Whether to resize inputs to the desired patch shape. 120 download: Whether to download the data if it is not present. 121 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 122 123 Returns: 124 The segmentation dataset. 125 """ 126 raw_paths, label_paths = get_bonedat_paths(path, download) 127 128 if resize_inputs: 129 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 130 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 131 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 132 ) 133 134 return torch_em.default_segmentation_dataset( 135 raw_paths=raw_paths, 136 raw_key="data", 137 label_paths=label_paths, 138 label_key="data", 139 patch_shape=patch_shape, 140 is_seg_dataset=True, 141 **kwargs 142 )
Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
145def get_bonedat_loader( 146 path: Union[os.PathLike, str], 147 batch_size: int, 148 patch_shape: Tuple[int, ...], 149 resize_inputs: bool = False, 150 download: bool = False, 151 **kwargs 152) -> DataLoader: 153 """Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation. 154 155 Args: 156 path: Filepath to a folder where the data is downloaded for further processing. 157 batch_size: The batch size for training. 158 patch_shape: The patch shape to use for training. 159 resize_inputs: Whether to resize inputs to the desired patch shape. 160 download: Whether to download the data if it is not present. 161 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 162 163 Returns: 164 The DataLoader. 165 """ 166 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 167 dataset = get_bonedat_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 168 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.