torch_em.data.datasets.medical.pansegdata
PanSegData contains annotations for pancreas segmentation in T1-weighted and T2-weighted abdominal MRI.
The dataset consists of 385 T1W and 382 T2W MRI volumes from five institutions, with binary pancreas labels.
NOTE: The dataset is distributed under the CC BY-NC 4.0 license. It is located at https://osf.io/kysnj/.
This dataset is from the publication https://doi.org/10.1016/j.media.2024.103382. Please cite it if you use this dataset in your research.
1"""PanSegData contains annotations for pancreas segmentation in T1-weighted and T2-weighted abdominal MRI. 2 3The dataset consists of 385 T1W and 382 T2W MRI volumes from five institutions, with binary pancreas labels. 4 5NOTE: The dataset is distributed under the CC BY-NC 4.0 license. It is located at https://osf.io/kysnj/. 6 7This dataset is from the publication https://doi.org/10.1016/j.media.2024.103382. 8Please cite it if you use this dataset in your research. 9""" 10 11import os 12from glob import glob 13from natsort import natsorted 14from typing import Union, Tuple, Literal, List 15 16from torch.utils.data import Dataset, DataLoader 17 18import torch_em 19 20from .. import util 21 22 23URLS = { 24 "t1": "https://osf.io/download/ch8ay/", 25 "t2": "https://osf.io/download/bre8p/", 26} 27 28CHECKSUMS = { 29 "t1": "f94c319f0ac8c627d1d871c542b4f8b2defcd206cb22959fd75965910f108b7f", 30 "t2": "c1b6ab676f92e27a3743a1bb9da7a3794a6c70075d2d4b724cc522bd6e6e1351", 31} 32 33 34def get_pansegdata_data(path: Union[os.PathLike, str], modality: Literal["t1", "t2"], download: bool = False) -> str: 35 """Download the PanSegData dataset. 36 37 Args: 38 path: Filepath to a folder where the data is downloaded for further processing. 39 modality: The choice of MRI modality. Either 't1' or 't2'. 40 download: Whether to download the data if it is not present. 41 42 Returns: 43 Filepath to the folder where the data is stored. 44 """ 45 if modality not in URLS: 46 raise ValueError(f"'{modality}' is not a valid modality. Choose either 't1' or 't2'.") 47 48 data_dir = os.path.join(path, modality) 49 if os.path.exists(data_dir): 50 return data_dir 51 52 os.makedirs(path, exist_ok=True) 53 54 zip_path = os.path.join(path, f"{modality}.zip") 55 util.download_source(path=zip_path, url=URLS[modality], download=download, checksum=CHECKSUMS[modality]) 56 util.unzip(zip_path=zip_path, dst=path) 57 58 return data_dir 59 60 61def get_pansegdata_paths( 62 path: Union[os.PathLike, str], modality: Literal["t1", "t2"] = "t2", download: bool = False 63) -> Tuple[List[str], List[str]]: 64 """Get paths to the PanSegData data. 65 66 Args: 67 path: Filepath to a folder where the data is downloaded for further processing. 68 modality: The choice of MRI modality. Either 't1' or 't2'. 69 download: Whether to download the data if it is not present. 70 71 Returns: 72 List of filepaths for the image data. 73 List of filepaths for the label data. 74 """ 75 data_dir = get_pansegdata_data(path, modality, download) 76 77 raw_paths = natsorted(glob(os.path.join(data_dir, "imagesTr", "*.nii.gz"))) 78 label_paths = natsorted(glob(os.path.join(data_dir, "labelsTr", "*.nii.gz"))) 79 80 assert len(raw_paths) > 0 and len(raw_paths) == len(label_paths) 81 82 return raw_paths, label_paths 83 84 85def get_pansegdata_dataset( 86 path: Union[os.PathLike, str], 87 patch_shape: Tuple[int, ...], 88 modality: Literal["t1", "t2"] = "t2", 89 resize_inputs: bool = False, 90 download: bool = False, 91 **kwargs 92) -> Dataset: 93 """Get the PanSegData dataset for pancreas segmentation. 94 95 Args: 96 path: Filepath to a folder where the data is downloaded for further processing. 97 patch_shape: The patch shape to use for training. 98 modality: The choice of MRI modality. Either 't1' or 't2'. 99 resize_inputs: Whether to resize inputs to the desired patch shape. 100 download: Whether to download the data if it is not present. 101 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 102 103 Returns: 104 The segmentation dataset. 105 """ 106 raw_paths, label_paths = get_pansegdata_paths(path, modality, download) 107 108 if resize_inputs: 109 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 110 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 111 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 112 ) 113 114 return torch_em.default_segmentation_dataset( 115 raw_paths=raw_paths, 116 raw_key="data", 117 label_paths=label_paths, 118 label_key="data", 119 patch_shape=patch_shape, 120 is_seg_dataset=True, 121 **kwargs 122 ) 123 124 125def get_pansegdata_loader( 126 path: Union[os.PathLike, str], 127 batch_size: int, 128 patch_shape: Tuple[int, ...], 129 modality: Literal["t1", "t2"] = "t2", 130 resize_inputs: bool = False, 131 download: bool = False, 132 **kwargs 133) -> DataLoader: 134 """Get the PanSegData dataloader for pancreas segmentation. 135 136 Args: 137 path: Filepath to a folder where the data is downloaded for further processing. 138 batch_size: The batch size for training. 139 patch_shape: The patch shape to use for training. 140 modality: The choice of MRI modality. Either 't1' or 't2'. 141 resize_inputs: Whether to resize inputs to the desired patch shape. 142 download: Whether to download the data if it is not present. 143 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 144 145 Returns: 146 The DataLoader. 147 """ 148 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 149 dataset = get_pansegdata_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 150 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
35def get_pansegdata_data(path: Union[os.PathLike, str], modality: Literal["t1", "t2"], download: bool = False) -> str: 36 """Download the PanSegData dataset. 37 38 Args: 39 path: Filepath to a folder where the data is downloaded for further processing. 40 modality: The choice of MRI modality. Either 't1' or 't2'. 41 download: Whether to download the data if it is not present. 42 43 Returns: 44 Filepath to the folder where the data is stored. 45 """ 46 if modality not in URLS: 47 raise ValueError(f"'{modality}' is not a valid modality. Choose either 't1' or 't2'.") 48 49 data_dir = os.path.join(path, modality) 50 if os.path.exists(data_dir): 51 return data_dir 52 53 os.makedirs(path, exist_ok=True) 54 55 zip_path = os.path.join(path, f"{modality}.zip") 56 util.download_source(path=zip_path, url=URLS[modality], download=download, checksum=CHECKSUMS[modality]) 57 util.unzip(zip_path=zip_path, dst=path) 58 59 return data_dir
Download the PanSegData dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The choice of MRI modality. Either 't1' or 't2'.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder where the data is stored.
62def get_pansegdata_paths( 63 path: Union[os.PathLike, str], modality: Literal["t1", "t2"] = "t2", download: bool = False 64) -> Tuple[List[str], List[str]]: 65 """Get paths to the PanSegData data. 66 67 Args: 68 path: Filepath to a folder where the data is downloaded for further processing. 69 modality: The choice of MRI modality. Either 't1' or 't2'. 70 download: Whether to download the data if it is not present. 71 72 Returns: 73 List of filepaths for the image data. 74 List of filepaths for the label data. 75 """ 76 data_dir = get_pansegdata_data(path, modality, download) 77 78 raw_paths = natsorted(glob(os.path.join(data_dir, "imagesTr", "*.nii.gz"))) 79 label_paths = natsorted(glob(os.path.join(data_dir, "labelsTr", "*.nii.gz"))) 80 81 assert len(raw_paths) > 0 and len(raw_paths) == len(label_paths) 82 83 return raw_paths, label_paths
Get paths to the PanSegData data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The choice of MRI modality. Either 't1' or 't2'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
86def get_pansegdata_dataset( 87 path: Union[os.PathLike, str], 88 patch_shape: Tuple[int, ...], 89 modality: Literal["t1", "t2"] = "t2", 90 resize_inputs: bool = False, 91 download: bool = False, 92 **kwargs 93) -> Dataset: 94 """Get the PanSegData dataset for pancreas segmentation. 95 96 Args: 97 path: Filepath to a folder where the data is downloaded for further processing. 98 patch_shape: The patch shape to use for training. 99 modality: The choice of MRI modality. Either 't1' or 't2'. 100 resize_inputs: Whether to resize inputs to the desired patch shape. 101 download: Whether to download the data if it is not present. 102 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 103 104 Returns: 105 The segmentation dataset. 106 """ 107 raw_paths, label_paths = get_pansegdata_paths(path, modality, download) 108 109 if resize_inputs: 110 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 111 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 112 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 113 ) 114 115 return torch_em.default_segmentation_dataset( 116 raw_paths=raw_paths, 117 raw_key="data", 118 label_paths=label_paths, 119 label_key="data", 120 patch_shape=patch_shape, 121 is_seg_dataset=True, 122 **kwargs 123 )
Get the PanSegData dataset for pancreas segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- modality: The choice of MRI modality. Either 't1' or 't2'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
126def get_pansegdata_loader( 127 path: Union[os.PathLike, str], 128 batch_size: int, 129 patch_shape: Tuple[int, ...], 130 modality: Literal["t1", "t2"] = "t2", 131 resize_inputs: bool = False, 132 download: bool = False, 133 **kwargs 134) -> DataLoader: 135 """Get the PanSegData dataloader for pancreas segmentation. 136 137 Args: 138 path: Filepath to a folder where the data is downloaded for further processing. 139 batch_size: The batch size for training. 140 patch_shape: The patch shape to use for training. 141 modality: The choice of MRI modality. Either 't1' or 't2'. 142 resize_inputs: Whether to resize inputs to the desired patch shape. 143 download: Whether to download the data if it is not present. 144 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 145 146 Returns: 147 The DataLoader. 148 """ 149 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 150 dataset = get_pansegdata_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 151 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the PanSegData dataloader for pancreas segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- modality: The choice of MRI modality. Either 't1' or 't2'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.