torch_em.data.datasets.medical.pansegdata

PanSegData contains annotations for pancreas segmentation in T1-weighted and T2-weighted abdominal MRI.

The dataset consists of 385 T1W and 382 T2W MRI volumes from five institutions, with binary pancreas labels.

NOTE: The dataset is distributed under the CC BY-NC 4.0 license. It is located at https://osf.io/kysnj/.

This dataset is from the publication https://doi.org/10.1016/j.media.2024.103382. Please cite it if you use this dataset in your research.

  1"""PanSegData contains annotations for pancreas segmentation in T1-weighted and T2-weighted abdominal MRI.
  2
  3The dataset consists of 385 T1W and 382 T2W MRI volumes from five institutions, with binary pancreas labels.
  4
  5NOTE: The dataset is distributed under the CC BY-NC 4.0 license. It is located at https://osf.io/kysnj/.
  6
  7This dataset is from the publication https://doi.org/10.1016/j.media.2024.103382.
  8Please cite it if you use this dataset in your research.
  9"""
 10
 11import os
 12from glob import glob
 13from natsort import natsorted
 14from typing import Union, Tuple, Literal, List
 15
 16from torch.utils.data import Dataset, DataLoader
 17
 18import torch_em
 19
 20from .. import util
 21
 22
 23URLS = {
 24    "t1": "https://osf.io/download/ch8ay/",
 25    "t2": "https://osf.io/download/bre8p/",
 26}
 27
 28CHECKSUMS = {
 29    "t1": "f94c319f0ac8c627d1d871c542b4f8b2defcd206cb22959fd75965910f108b7f",
 30    "t2": "c1b6ab676f92e27a3743a1bb9da7a3794a6c70075d2d4b724cc522bd6e6e1351",
 31}
 32
 33
 34def get_pansegdata_data(path: Union[os.PathLike, str], modality: Literal["t1", "t2"], download: bool = False) -> str:
 35    """Download the PanSegData dataset.
 36
 37    Args:
 38        path: Filepath to a folder where the data is downloaded for further processing.
 39        modality: The choice of MRI modality. Either 't1' or 't2'.
 40        download: Whether to download the data if it is not present.
 41
 42    Returns:
 43        Filepath to the folder where the data is stored.
 44    """
 45    if modality not in URLS:
 46        raise ValueError(f"'{modality}' is not a valid modality. Choose either 't1' or 't2'.")
 47
 48    data_dir = os.path.join(path, modality)
 49    if os.path.exists(data_dir):
 50        return data_dir
 51
 52    os.makedirs(path, exist_ok=True)
 53
 54    zip_path = os.path.join(path, f"{modality}.zip")
 55    util.download_source(path=zip_path, url=URLS[modality], download=download, checksum=CHECKSUMS[modality])
 56    util.unzip(zip_path=zip_path, dst=path)
 57
 58    return data_dir
 59
 60
 61def get_pansegdata_paths(
 62    path: Union[os.PathLike, str], modality: Literal["t1", "t2"] = "t2", download: bool = False
 63) -> Tuple[List[str], List[str]]:
 64    """Get paths to the PanSegData data.
 65
 66    Args:
 67        path: Filepath to a folder where the data is downloaded for further processing.
 68        modality: The choice of MRI modality. Either 't1' or 't2'.
 69        download: Whether to download the data if it is not present.
 70
 71    Returns:
 72        List of filepaths for the image data.
 73        List of filepaths for the label data.
 74    """
 75    data_dir = get_pansegdata_data(path, modality, download)
 76
 77    raw_paths = natsorted(glob(os.path.join(data_dir, "imagesTr", "*.nii.gz")))
 78    label_paths = natsorted(glob(os.path.join(data_dir, "labelsTr", "*.nii.gz")))
 79
 80    assert len(raw_paths) > 0 and len(raw_paths) == len(label_paths)
 81
 82    return raw_paths, label_paths
 83
 84
 85def get_pansegdata_dataset(
 86    path: Union[os.PathLike, str],
 87    patch_shape: Tuple[int, ...],
 88    modality: Literal["t1", "t2"] = "t2",
 89    resize_inputs: bool = False,
 90    download: bool = False,
 91    **kwargs
 92) -> Dataset:
 93    """Get the PanSegData dataset for pancreas segmentation.
 94
 95    Args:
 96        path: Filepath to a folder where the data is downloaded for further processing.
 97        patch_shape: The patch shape to use for training.
 98        modality: The choice of MRI modality. Either 't1' or 't2'.
 99        resize_inputs: Whether to resize inputs to the desired patch shape.
100        download: Whether to download the data if it is not present.
101        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
102
103    Returns:
104        The segmentation dataset.
105    """
106    raw_paths, label_paths = get_pansegdata_paths(path, modality, download)
107
108    if resize_inputs:
109        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
110        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
111            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
112        )
113
114    return torch_em.default_segmentation_dataset(
115        raw_paths=raw_paths,
116        raw_key="data",
117        label_paths=label_paths,
118        label_key="data",
119        patch_shape=patch_shape,
120        is_seg_dataset=True,
121        **kwargs
122    )
123
124
125def get_pansegdata_loader(
126    path: Union[os.PathLike, str],
127    batch_size: int,
128    patch_shape: Tuple[int, ...],
129    modality: Literal["t1", "t2"] = "t2",
130    resize_inputs: bool = False,
131    download: bool = False,
132    **kwargs
133) -> DataLoader:
134    """Get the PanSegData dataloader for pancreas segmentation.
135
136    Args:
137        path: Filepath to a folder where the data is downloaded for further processing.
138        batch_size: The batch size for training.
139        patch_shape: The patch shape to use for training.
140        modality: The choice of MRI modality. Either 't1' or 't2'.
141        resize_inputs: Whether to resize inputs to the desired patch shape.
142        download: Whether to download the data if it is not present.
143        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
144
145    Returns:
146        The DataLoader.
147    """
148    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
149    dataset = get_pansegdata_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
150    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'t1': 'https://osf.io/download/ch8ay/', 't2': 'https://osf.io/download/bre8p/'}
CHECKSUMS = {'t1': 'f94c319f0ac8c627d1d871c542b4f8b2defcd206cb22959fd75965910f108b7f', 't2': 'c1b6ab676f92e27a3743a1bb9da7a3794a6c70075d2d4b724cc522bd6e6e1351'}
def get_pansegdata_data( path: Union[os.PathLike, str], modality: Literal['t1', 't2'], download: bool = False) -> str:
35def get_pansegdata_data(path: Union[os.PathLike, str], modality: Literal["t1", "t2"], download: bool = False) -> str:
36    """Download the PanSegData dataset.
37
38    Args:
39        path: Filepath to a folder where the data is downloaded for further processing.
40        modality: The choice of MRI modality. Either 't1' or 't2'.
41        download: Whether to download the data if it is not present.
42
43    Returns:
44        Filepath to the folder where the data is stored.
45    """
46    if modality not in URLS:
47        raise ValueError(f"'{modality}' is not a valid modality. Choose either 't1' or 't2'.")
48
49    data_dir = os.path.join(path, modality)
50    if os.path.exists(data_dir):
51        return data_dir
52
53    os.makedirs(path, exist_ok=True)
54
55    zip_path = os.path.join(path, f"{modality}.zip")
56    util.download_source(path=zip_path, url=URLS[modality], download=download, checksum=CHECKSUMS[modality])
57    util.unzip(zip_path=zip_path, dst=path)
58
59    return data_dir

Download the PanSegData dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of MRI modality. Either 't1' or 't2'.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder where the data is stored.

def get_pansegdata_paths( path: Union[os.PathLike, str], modality: Literal['t1', 't2'] = 't2', download: bool = False) -> Tuple[List[str], List[str]]:
62def get_pansegdata_paths(
63    path: Union[os.PathLike, str], modality: Literal["t1", "t2"] = "t2", download: bool = False
64) -> Tuple[List[str], List[str]]:
65    """Get paths to the PanSegData data.
66
67    Args:
68        path: Filepath to a folder where the data is downloaded for further processing.
69        modality: The choice of MRI modality. Either 't1' or 't2'.
70        download: Whether to download the data if it is not present.
71
72    Returns:
73        List of filepaths for the image data.
74        List of filepaths for the label data.
75    """
76    data_dir = get_pansegdata_data(path, modality, download)
77
78    raw_paths = natsorted(glob(os.path.join(data_dir, "imagesTr", "*.nii.gz")))
79    label_paths = natsorted(glob(os.path.join(data_dir, "labelsTr", "*.nii.gz")))
80
81    assert len(raw_paths) > 0 and len(raw_paths) == len(label_paths)
82
83    return raw_paths, label_paths

Get paths to the PanSegData data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of MRI modality. Either 't1' or 't2'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_pansegdata_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], modality: Literal['t1', 't2'] = 't2', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 86def get_pansegdata_dataset(
 87    path: Union[os.PathLike, str],
 88    patch_shape: Tuple[int, ...],
 89    modality: Literal["t1", "t2"] = "t2",
 90    resize_inputs: bool = False,
 91    download: bool = False,
 92    **kwargs
 93) -> Dataset:
 94    """Get the PanSegData dataset for pancreas segmentation.
 95
 96    Args:
 97        path: Filepath to a folder where the data is downloaded for further processing.
 98        patch_shape: The patch shape to use for training.
 99        modality: The choice of MRI modality. Either 't1' or 't2'.
100        resize_inputs: Whether to resize inputs to the desired patch shape.
101        download: Whether to download the data if it is not present.
102        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
103
104    Returns:
105        The segmentation dataset.
106    """
107    raw_paths, label_paths = get_pansegdata_paths(path, modality, download)
108
109    if resize_inputs:
110        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
111        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
112            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
113        )
114
115    return torch_em.default_segmentation_dataset(
116        raw_paths=raw_paths,
117        raw_key="data",
118        label_paths=label_paths,
119        label_key="data",
120        patch_shape=patch_shape,
121        is_seg_dataset=True,
122        **kwargs
123    )

Get the PanSegData dataset for pancreas segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of MRI modality. Either 't1' or 't2'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pansegdata_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], modality: Literal['t1', 't2'] = 't2', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
126def get_pansegdata_loader(
127    path: Union[os.PathLike, str],
128    batch_size: int,
129    patch_shape: Tuple[int, ...],
130    modality: Literal["t1", "t2"] = "t2",
131    resize_inputs: bool = False,
132    download: bool = False,
133    **kwargs
134) -> DataLoader:
135    """Get the PanSegData dataloader for pancreas segmentation.
136
137    Args:
138        path: Filepath to a folder where the data is downloaded for further processing.
139        batch_size: The batch size for training.
140        patch_shape: The patch shape to use for training.
141        modality: The choice of MRI modality. Either 't1' or 't2'.
142        resize_inputs: Whether to resize inputs to the desired patch shape.
143        download: Whether to download the data if it is not present.
144        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
145
146    Returns:
147        The DataLoader.
148    """
149    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
150    dataset = get_pansegdata_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
151    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the PanSegData dataloader for pancreas segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of MRI modality. Either 't1' or 't2'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.