torch_em.data.datasets.medical.lumvbcanseg

The LumVBCanSeg dataset contains annotations for lumbar vertebral body cancellous bone segmentation in CT.

The dataset consists of 185 lumbar CT scans acquired at ShengJing Hospital of China Medical University on multiple Philips and Siemens scanners, resampled to an isotropic 1x1x1 mm resolution. Cases with vertebral fractures, metal implants, bone tumors or foreign materials were excluded. Each scan has a corresponding voxel-level segmentation mask covering the cancellous bone of the five lumbar vertebral bodies (L1-L5), with labels 1-5 respectively. Annotations were made by 3 physicians and refined, dismissed or approved by a physician with more than 30 years of experience in lumbar imaging.

The data is located at https://doi.org/10.5281/zenodo.8181250, released under a CC-BY-4.0 license. NOTE: The archive is a single ~9.4 GB zip file with a flat 'Task908' folder, pairing each raw volume ('.nii.gz') with its mask ('_seg.nii.gz').

This dataset is from the publication https://doi.org/10.1016/j.compbiomed.2024.108237. Please cite it if you use this dataset for your research.

  1"""The LumVBCanSeg dataset contains annotations for lumbar vertebral body cancellous bone segmentation in CT.
  2
  3The dataset consists of 185 lumbar CT scans acquired at ShengJing Hospital of China Medical University on
  4multiple Philips and Siemens scanners, resampled to an isotropic 1x1x1 mm resolution. Cases with vertebral
  5fractures, metal implants, bone tumors or foreign materials were excluded. Each scan has a corresponding
  6voxel-level segmentation mask covering the cancellous bone of the five lumbar vertebral bodies (L1-L5), with
  7labels 1-5 respectively. Annotations were made by 3 physicians and refined, dismissed or approved by a
  8physician with more than 30 years of experience in lumbar imaging.
  9
 10The data is located at https://doi.org/10.5281/zenodo.8181250, released under a CC-BY-4.0 license.
 11NOTE: The archive is a single ~9.4 GB zip file with a flat 'Task908' folder, pairing each raw volume
 12('<case_id>.nii.gz') with its mask ('<case_id>_seg.nii.gz').
 13
 14This dataset is from the publication https://doi.org/10.1016/j.compbiomed.2024.108237.
 15Please cite it if you use this dataset for your research.
 16"""
 17
 18import os
 19from glob import glob
 20from natsort import natsorted
 21from typing import Union, Tuple, List
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30URL = "https://zenodo.org/api/records/8181250/files/LumVBCanSeg.zip/content"
 31CHECKSUM = "e41f1f5bae9611494997ac6f5f9d92e36ea80c4d6c417f98752932384f75f846"
 32
 33
 34def get_lumvbcanseg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 35    """Download the LumVBCanSeg dataset.
 36
 37    Args:
 38        path: Filepath to a folder where the data is downloaded for further processing.
 39        download: Whether to download the data if it is not present.
 40
 41    Returns:
 42        Filepath where the data is downloaded.
 43    """
 44    data_dir = os.path.join(path, "Task908")
 45    if os.path.exists(data_dir):
 46        return data_dir
 47
 48    os.makedirs(path, exist_ok=True)
 49
 50    zip_path = os.path.join(path, "LumVBCanSeg.zip")
 51    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 52    util.unzip(zip_path=zip_path, dst=path)
 53
 54    return data_dir
 55
 56
 57def get_lumvbcanseg_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 58    """Get paths to the LumVBCanSeg data.
 59
 60    Args:
 61        path: Filepath to a folder where the data is downloaded for further processing.
 62        download: Whether to download the data if it is not present.
 63
 64    Returns:
 65        List of filepaths for the image data.
 66        List of filepaths for the label data.
 67    """
 68    data_dir = get_lumvbcanseg_data(path, download)
 69
 70    label_paths = natsorted(glob(os.path.join(data_dir, "*_seg.nii.gz")))
 71    raw_paths = [p.replace("_seg.nii.gz", ".nii.gz") for p in label_paths]
 72
 73    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 74    assert all(os.path.exists(p) for p in raw_paths)
 75
 76    return raw_paths, label_paths
 77
 78
 79def get_lumvbcanseg_dataset(
 80    path: Union[os.PathLike, str],
 81    patch_shape: Tuple[int, int, int],
 82    resize_inputs: bool = False,
 83    download: bool = False,
 84    **kwargs
 85) -> Dataset:
 86    """Get the LumVBCanSeg dataset for lumbar vertebral body cancellous bone segmentation.
 87
 88    Args:
 89        path: Filepath to a folder where the data is downloaded for further processing.
 90        patch_shape: The patch shape to use for training.
 91        resize_inputs: Whether to resize the inputs to the patch shape.
 92        download: Whether to download the data if it is not present.
 93        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 94
 95    Returns:
 96        The segmentation dataset.
 97    """
 98    raw_paths, label_paths = get_lumvbcanseg_paths(path, download)
 99
100    if resize_inputs:
101        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
102        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
103            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
104        )
105
106    return torch_em.default_segmentation_dataset(
107        raw_paths=raw_paths,
108        raw_key="data",
109        label_paths=label_paths,
110        label_key="data",
111        patch_shape=patch_shape,
112        ndim=3,
113        **kwargs
114    )
115
116
117def get_lumvbcanseg_loader(
118    path: Union[os.PathLike, str],
119    batch_size: int,
120    patch_shape: Tuple[int, int, int],
121    resize_inputs: bool = False,
122    download: bool = False,
123    **kwargs
124) -> DataLoader:
125    """Get the LumVBCanSeg dataloader for lumbar vertebral body cancellous bone segmentation.
126
127    Args:
128        path: Filepath to a folder where the data is downloaded for further processing.
129        batch_size: The batch size for training.
130        patch_shape: The patch shape to use for training.
131        resize_inputs: Whether to resize the inputs to the patch shape.
132        download: Whether to download the data if it is not present.
133        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
134
135    Returns:
136        The DataLoader.
137    """
138    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
139    dataset = get_lumvbcanseg_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
140    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/api/records/8181250/files/LumVBCanSeg.zip/content'
CHECKSUM = 'e41f1f5bae9611494997ac6f5f9d92e36ea80c4d6c417f98752932384f75f846'
def get_lumvbcanseg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
35def get_lumvbcanseg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
36    """Download the LumVBCanSeg dataset.
37
38    Args:
39        path: Filepath to a folder where the data is downloaded for further processing.
40        download: Whether to download the data if it is not present.
41
42    Returns:
43        Filepath where the data is downloaded.
44    """
45    data_dir = os.path.join(path, "Task908")
46    if os.path.exists(data_dir):
47        return data_dir
48
49    os.makedirs(path, exist_ok=True)
50
51    zip_path = os.path.join(path, "LumVBCanSeg.zip")
52    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
53    util.unzip(zip_path=zip_path, dst=path)
54
55    return data_dir

Download the LumVBCanSeg dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_lumvbcanseg_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
58def get_lumvbcanseg_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
59    """Get paths to the LumVBCanSeg data.
60
61    Args:
62        path: Filepath to a folder where the data is downloaded for further processing.
63        download: Whether to download the data if it is not present.
64
65    Returns:
66        List of filepaths for the image data.
67        List of filepaths for the label data.
68    """
69    data_dir = get_lumvbcanseg_data(path, download)
70
71    label_paths = natsorted(glob(os.path.join(data_dir, "*_seg.nii.gz")))
72    raw_paths = [p.replace("_seg.nii.gz", ".nii.gz") for p in label_paths]
73
74    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
75    assert all(os.path.exists(p) for p in raw_paths)
76
77    return raw_paths, label_paths

Get paths to the LumVBCanSeg data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_lumvbcanseg_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 80def get_lumvbcanseg_dataset(
 81    path: Union[os.PathLike, str],
 82    patch_shape: Tuple[int, int, int],
 83    resize_inputs: bool = False,
 84    download: bool = False,
 85    **kwargs
 86) -> Dataset:
 87    """Get the LumVBCanSeg dataset for lumbar vertebral body cancellous bone segmentation.
 88
 89    Args:
 90        path: Filepath to a folder where the data is downloaded for further processing.
 91        patch_shape: The patch shape to use for training.
 92        resize_inputs: Whether to resize the inputs to the patch shape.
 93        download: Whether to download the data if it is not present.
 94        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 95
 96    Returns:
 97        The segmentation dataset.
 98    """
 99    raw_paths, label_paths = get_lumvbcanseg_paths(path, download)
100
101    if resize_inputs:
102        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
103        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
104            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
105        )
106
107    return torch_em.default_segmentation_dataset(
108        raw_paths=raw_paths,
109        raw_key="data",
110        label_paths=label_paths,
111        label_key="data",
112        patch_shape=patch_shape,
113        ndim=3,
114        **kwargs
115    )

Get the LumVBCanSeg dataset for lumbar vertebral body cancellous bone segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_lumvbcanseg_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
118def get_lumvbcanseg_loader(
119    path: Union[os.PathLike, str],
120    batch_size: int,
121    patch_shape: Tuple[int, int, int],
122    resize_inputs: bool = False,
123    download: bool = False,
124    **kwargs
125) -> DataLoader:
126    """Get the LumVBCanSeg dataloader for lumbar vertebral body cancellous bone segmentation.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        batch_size: The batch size for training.
131        patch_shape: The patch shape to use for training.
132        resize_inputs: Whether to resize the inputs to the patch shape.
133        download: Whether to download the data if it is not present.
134        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
135
136    Returns:
137        The DataLoader.
138    """
139    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
140    dataset = get_lumvbcanseg_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
141    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the LumVBCanSeg dataloader for lumbar vertebral body cancellous bone segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.