torch_em.data.datasets.medical.bonedat

BoneDat is a dataset for the segmentation of the pelvis, sacrum and L4/L5 vertebrae in lumbopelvic CT.

The dataset consists of 278 anonymized human lumbopelvic CT scans (ages 16-91, balanced by sex), together with multi-label segmentation masks that delineate the individual pelvic bones, the sacrum and the L4/L5 vertebrae. The masks were generated with the Biomedisa algorithm and refined and validated by experts.

The dataset is hosted on Zenodo as 8 linked records. The main record (https://doi.org/10.5281/zenodo.15189761) only holds supplementary results and metadata from the associated publication, not usable image data. The raw CT scans and segmentation masks are distributed across 7 records that together form a single split rar archive (BoneDat.part1.rar - BoneDat.part7.rar, https://doi.org/10.5281/zenodo.15188359 and follow-up DOIs). Most of this archive (parts 1-6, about 290 GB) holds registration and template outputs (deformation fields, warped volumes, meshes) that are not exposed by this module. The raw CT scans and segmentation masks used here are both fully contained within part 7 alone (about 27 GB), so this module only downloads that part.

All 8 records are licensed under CC-BY-4.0. Note that the article text itself is under a separate, more restrictive Springer Nature license, but this does not apply to the dataset files, which are CC-BY-4.0.

This dataset is from the publication https://doi.org/10.1038/s41597-025-05161-y. Please cite it if you use this dataset for your research.

  1"""BoneDat is a dataset for the segmentation of the pelvis, sacrum and L4/L5 vertebrae in lumbopelvic CT.
  2
  3The dataset consists of 278 anonymized human lumbopelvic CT scans (ages 16-91, balanced by sex), together with
  4multi-label segmentation masks that delineate the individual pelvic bones, the sacrum and the L4/L5 vertebrae.
  5The masks were generated with the Biomedisa algorithm and refined and validated by experts.
  6
  7The dataset is hosted on Zenodo as 8 linked records. The main record (https://doi.org/10.5281/zenodo.15189761)
  8only holds supplementary results and metadata from the associated publication, not usable image data. The raw
  9CT scans and segmentation masks are distributed across 7 records that together form a single split rar archive
 10(BoneDat.part1.rar - BoneDat.part7.rar, https://doi.org/10.5281/zenodo.15188359 and follow-up DOIs). Most of
 11this archive (parts 1-6, about 290 GB) holds registration and template outputs (deformation fields, warped
 12volumes, meshes) that are not exposed by this module. The raw CT scans and segmentation masks used here are
 13both fully contained within part 7 alone (about 27 GB), so this module only downloads that part.
 14
 15All 8 records are licensed under CC-BY-4.0. Note that the article text itself is under a separate, more
 16restrictive Springer Nature license, but this does not apply to the dataset files, which are CC-BY-4.0.
 17
 18This dataset is from the publication https://doi.org/10.1038/s41597-025-05161-y. Please cite it if you use this
 19dataset for your research.
 20"""
 21
 22import os
 23import shutil
 24import subprocess
 25from glob import glob
 26from natsort import natsorted
 27from typing import Union, Tuple, List
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36URL = "https://zenodo.org/records/15189605/files/BoneDat.part7.rar?download=1"
 37CHECKSUM = "f27bf0813df07c1cceff822246e2793bb03ca74e10f6e7e87861e19b9bdafb85"
 38
 39
 40def _extract_bonedat(rar_path, data_dir):
 41    if shutil.which("7z") is None:
 42        raise RuntimeError(
 43            "Need the 'p7zip' CLI to extract this archive. You can install it via 'conda install -c conda-forge p7zip'."  # noqa
 44        )
 45
 46    # The rar archive is part of a 7-part split archive (this is part 7). Extracting it in isolation (without
 47    # the preceding 6 parts) makes '7z' report a non-fatal 'Headers Error' for the missing volume chain, but the
 48    # files that are fully contained within this part (i.e. everything under 'raw/' and 'derived/segmentation/')
 49    # are still extracted correctly. We therefore do not check the subprocess return code here and instead
 50    # verify the presence of the expected output further below.
 51    subprocess.run(["7z", "x", f"-o{data_dir}", "-y", rar_path, "raw/*", "derived/segmentation/*"])
 52
 53
 54def get_bonedat_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 55    """Download the BoneDat dataset.
 56
 57    Args:
 58        path: Filepath to a folder where the data is downloaded for further processing.
 59        download: Whether to download the data if it is not present.
 60
 61    Returns:
 62        Filepath where the data is downloaded.
 63    """
 64    data_dir = os.path.join(path, "data")
 65    if os.path.exists(os.path.join(data_dir, "raw")) and os.path.exists(os.path.join(data_dir, "derived")):
 66        return data_dir
 67
 68    os.makedirs(path, exist_ok=True)
 69
 70    rar_path = os.path.join(path, "BoneDat.part7.rar")
 71    util.download_source(path=rar_path, url=URL, download=download, checksum=CHECKSUM)
 72
 73    os.makedirs(data_dir, exist_ok=True)
 74    _extract_bonedat(rar_path, data_dir)
 75
 76    if not os.path.exists(os.path.join(data_dir, "raw")) or not os.path.exists(os.path.join(data_dir, "derived")):
 77        raise RuntimeError(f"Extraction seems to have failed: could not find the expected data at '{data_dir}'.")
 78
 79    os.remove(rar_path)
 80
 81    return data_dir
 82
 83
 84def get_bonedat_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 85    """Get paths to the BoneDat data.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        List of filepaths for the image data.
 93        List of filepaths for the label data.
 94    """
 95    data_dir = get_bonedat_data(path, download)
 96
 97    raw_paths = natsorted(glob(os.path.join(data_dir, "raw", "*", "original.nii.gz")))
 98    label_paths = natsorted(glob(os.path.join(data_dir, "derived", "segmentation", "*", "mask.nii.gz")))
 99
100    if len(raw_paths) == 0 or len(raw_paths) != len(label_paths):
101        raise RuntimeError("Something went wrong with fetching the image and label paths.")
102
103    return raw_paths, label_paths
104
105
106def get_bonedat_dataset(
107    path: Union[os.PathLike, str],
108    patch_shape: Tuple[int, ...],
109    resize_inputs: bool = False,
110    download: bool = False,
111    **kwargs
112) -> Dataset:
113    """Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation.
114
115    Args:
116        path: Filepath to a folder where the data is downloaded for further processing.
117        patch_shape: The patch shape to use for training.
118        resize_inputs: Whether to resize inputs to the desired patch shape.
119        download: Whether to download the data if it is not present.
120        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
121
122    Returns:
123        The segmentation dataset.
124    """
125    raw_paths, label_paths = get_bonedat_paths(path, download)
126
127    if resize_inputs:
128        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
129        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
130            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
131        )
132
133    return torch_em.default_segmentation_dataset(
134        raw_paths=raw_paths,
135        raw_key="data",
136        label_paths=label_paths,
137        label_key="data",
138        patch_shape=patch_shape,
139        is_seg_dataset=True,
140        **kwargs
141    )
142
143
144def get_bonedat_loader(
145    path: Union[os.PathLike, str],
146    batch_size: int,
147    patch_shape: Tuple[int, ...],
148    resize_inputs: bool = False,
149    download: bool = False,
150    **kwargs
151) -> DataLoader:
152    """Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation.
153
154    Args:
155        path: Filepath to a folder where the data is downloaded for further processing.
156        batch_size: The batch size for training.
157        patch_shape: The patch shape to use for training.
158        resize_inputs: Whether to resize inputs to the desired patch shape.
159        download: Whether to download the data if it is not present.
160        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
161
162    Returns:
163        The DataLoader.
164    """
165    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
166    dataset = get_bonedat_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
167    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/15189605/files/BoneDat.part7.rar?download=1'
CHECKSUM = 'f27bf0813df07c1cceff822246e2793bb03ca74e10f6e7e87861e19b9bdafb85'
def get_bonedat_data(path: Union[os.PathLike, str], download: bool = False) -> str:
55def get_bonedat_data(path: Union[os.PathLike, str], download: bool = False) -> str:
56    """Download the BoneDat dataset.
57
58    Args:
59        path: Filepath to a folder where the data is downloaded for further processing.
60        download: Whether to download the data if it is not present.
61
62    Returns:
63        Filepath where the data is downloaded.
64    """
65    data_dir = os.path.join(path, "data")
66    if os.path.exists(os.path.join(data_dir, "raw")) and os.path.exists(os.path.join(data_dir, "derived")):
67        return data_dir
68
69    os.makedirs(path, exist_ok=True)
70
71    rar_path = os.path.join(path, "BoneDat.part7.rar")
72    util.download_source(path=rar_path, url=URL, download=download, checksum=CHECKSUM)
73
74    os.makedirs(data_dir, exist_ok=True)
75    _extract_bonedat(rar_path, data_dir)
76
77    if not os.path.exists(os.path.join(data_dir, "raw")) or not os.path.exists(os.path.join(data_dir, "derived")):
78        raise RuntimeError(f"Extraction seems to have failed: could not find the expected data at '{data_dir}'.")
79
80    os.remove(rar_path)
81
82    return data_dir

Download the BoneDat dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_bonedat_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 85def get_bonedat_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 86    """Get paths to the BoneDat data.
 87
 88    Args:
 89        path: Filepath to a folder where the data is downloaded for further processing.
 90        download: Whether to download the data if it is not present.
 91
 92    Returns:
 93        List of filepaths for the image data.
 94        List of filepaths for the label data.
 95    """
 96    data_dir = get_bonedat_data(path, download)
 97
 98    raw_paths = natsorted(glob(os.path.join(data_dir, "raw", "*", "original.nii.gz")))
 99    label_paths = natsorted(glob(os.path.join(data_dir, "derived", "segmentation", "*", "mask.nii.gz")))
100
101    if len(raw_paths) == 0 or len(raw_paths) != len(label_paths):
102        raise RuntimeError("Something went wrong with fetching the image and label paths.")
103
104    return raw_paths, label_paths

Get paths to the BoneDat data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_bonedat_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
107def get_bonedat_dataset(
108    path: Union[os.PathLike, str],
109    patch_shape: Tuple[int, ...],
110    resize_inputs: bool = False,
111    download: bool = False,
112    **kwargs
113) -> Dataset:
114    """Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation.
115
116    Args:
117        path: Filepath to a folder where the data is downloaded for further processing.
118        patch_shape: The patch shape to use for training.
119        resize_inputs: Whether to resize inputs to the desired patch shape.
120        download: Whether to download the data if it is not present.
121        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
122
123    Returns:
124        The segmentation dataset.
125    """
126    raw_paths, label_paths = get_bonedat_paths(path, download)
127
128    if resize_inputs:
129        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
130        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
131            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
132        )
133
134    return torch_em.default_segmentation_dataset(
135        raw_paths=raw_paths,
136        raw_key="data",
137        label_paths=label_paths,
138        label_key="data",
139        patch_shape=patch_shape,
140        is_seg_dataset=True,
141        **kwargs
142    )

Get the BoneDat dataset for pelvis, sacrum and L4/L5 vertebra segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bonedat_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
145def get_bonedat_loader(
146    path: Union[os.PathLike, str],
147    batch_size: int,
148    patch_shape: Tuple[int, ...],
149    resize_inputs: bool = False,
150    download: bool = False,
151    **kwargs
152) -> DataLoader:
153    """Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation.
154
155    Args:
156        path: Filepath to a folder where the data is downloaded for further processing.
157        batch_size: The batch size for training.
158        patch_shape: The patch shape to use for training.
159        resize_inputs: Whether to resize inputs to the desired patch shape.
160        download: Whether to download the data if it is not present.
161        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
162
163    Returns:
164        The DataLoader.
165    """
166    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
167    dataset = get_bonedat_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
168    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BoneDat dataloader for pelvis, sacrum and L4/L5 vertebra segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.