torch_em.data.datasets.medical.figshare_brain_tumor
The Figshare Brain Tumor dataset contains annotations for meningioma, glioma and pituitary tumor segmentation in T1-weighted contrast-enhanced brain MRI.
The dataset consists of 3064 slices from 233 patients, acquired at Nanfang Hospital, Guangzhou, China
and General Hospital, Tianjin Medical University, China from 2005 to 2010, with an in-plane resolution
of 512 x 512 and pixel size of 0.49 x 0.49 mm^2. Each slice comes with a tumor type label (meningioma:
708 slices, glioma: 1426 slices, pituitary tumor: 930 slices) and a binary tumor mask, both distributed
as v7.3 matlab '.mat' files (readable with h5py, since scipy.io.loadmat cannot read this format).
The dataset is located at https://doi.org/10.6084/m9.figshare.1512427 and is distributed under the CC BY 4.0 license.
This dataset is from the publications https://doi.org/10.1371/journal.pone.0140381 and https://doi.org/10.1371/journal.pone.0157112. Please cite them if you use this dataset in your research.
1"""The Figshare Brain Tumor dataset contains annotations for meningioma, glioma and pituitary tumor 2segmentation in T1-weighted contrast-enhanced brain MRI. 3 4The dataset consists of 3064 slices from 233 patients, acquired at Nanfang Hospital, Guangzhou, China 5and General Hospital, Tianjin Medical University, China from 2005 to 2010, with an in-plane resolution 6of 512 x 512 and pixel size of 0.49 x 0.49 mm^2. Each slice comes with a tumor type label (meningioma: 7708 slices, glioma: 1426 slices, pituitary tumor: 930 slices) and a binary tumor mask, both distributed 8as v7.3 matlab '.mat' files (readable with `h5py`, since `scipy.io.loadmat` cannot read this format). 9 10The dataset is located at https://doi.org/10.6084/m9.figshare.1512427 and is distributed under the 11CC BY 4.0 license. 12 13This dataset is from the publications https://doi.org/10.1371/journal.pone.0140381 and 14https://doi.org/10.1371/journal.pone.0157112. Please cite them if you use this dataset in your research. 15""" 16 17import os 18import shutil 19from glob import glob 20from tqdm import tqdm 21from natsort import natsorted 22from typing import Union, Tuple, List, Optional 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31URLS = { 32 "brainTumorDataPublic_1-766.zip": "https://ndownloader.figshare.com/files/3381290", 33 "brainTumorDataPublic_767-1532.zip": "https://ndownloader.figshare.com/files/3381296", 34 "brainTumorDataPublic_1533-2298.zip": "https://ndownloader.figshare.com/files/3381293", 35 "brainTumorDataPublic_2299-3064.zip": "https://ndownloader.figshare.com/files/3381302", 36} 37 38CHECKSUMS = { 39 "brainTumorDataPublic_1-766.zip": "cdac0a7f4152cb34dd79b13543da96c37a6928ad8cb823e2662150b1a73ade54", 40 "brainTumorDataPublic_767-1532.zip": "d18e896d14f7af791cdd3d9b9342f9f974742a06132effde5a5219209862b4ce", 41 "brainTumorDataPublic_1533-2298.zip": "612d9f506c55c54db5f2f38b6d50741ef97a6d7c32b3e1be76e789674a97aefb", 42 "brainTumorDataPublic_2299-3064.zip": "9a673c55d139133cfbc57097bf155b5ac9f33684ee97f40bf2afba332d416a3a", 43} 44 45TUMOR_TYPES = {1: "meningioma", 2: "glioma", 3: "pituitary"} 46 47 48def _preprocess_figshare_brain_tumor(mat_dir, preprocessed_dir): 49 import h5py 50 51 os.makedirs(preprocessed_dir, exist_ok=True) 52 mat_paths = natsorted(glob(os.path.join(mat_dir, "*.mat"))) 53 for mat_path in tqdm(mat_paths, desc="Preprocessing inputs"): 54 sample_id = os.path.splitext(os.path.basename(mat_path))[0] 55 56 with h5py.File(mat_path, "r") as f: 57 cjdata = f["cjdata"] 58 image = cjdata["image"][:] 59 mask = cjdata["tumorMask"][:] 60 label = int(cjdata["label"][()].item()) 61 62 out_path = os.path.join(preprocessed_dir, f"{sample_id}_{TUMOR_TYPES[label]}.h5") 63 with h5py.File(out_path, "w") as f: 64 f.create_dataset("raw", data=image, compression="gzip") 65 f.create_dataset("labels", data=mask, compression="gzip") 66 67 shutil.rmtree(mat_dir) 68 69 70def get_figshare_brain_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str: 71 """Download the Figshare Brain Tumor data. 72 73 Args: 74 path: Filepath to a folder where the data is downloaded for further processing. 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 Filepath where the preprocessed data is stored. 79 """ 80 preprocessed_dir = os.path.join(path, "preprocessed") 81 if os.path.exists(preprocessed_dir) and glob(os.path.join(preprocessed_dir, "*.h5")): 82 return preprocessed_dir 83 84 os.makedirs(path, exist_ok=True) 85 86 mat_dir = os.path.join(path, "mat_files") 87 os.makedirs(mat_dir, exist_ok=True) 88 for fname, url in URLS.items(): 89 zip_path = os.path.join(path, fname) 90 util.download_source(path=zip_path, url=url, download=download, checksum=CHECKSUMS[fname]) 91 util.unzip(zip_path=zip_path, dst=mat_dir) 92 93 _preprocess_figshare_brain_tumor(mat_dir, preprocessed_dir) 94 return preprocessed_dir 95 96 97def get_figshare_brain_tumor_paths( 98 path: Union[os.PathLike, str], tumor_type: Optional[str] = None, download: bool = False 99) -> List[str]: 100 """Get paths to the Figshare Brain Tumor data. 101 102 Args: 103 path: Filepath to a folder where the data is downloaded for further processing. 104 tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'. 105 download: Whether to download the data if it is not present. 106 107 Returns: 108 List of filepaths for the stored data. 109 """ 110 preprocessed_dir = get_figshare_brain_tumor_data(path, download) 111 112 if tumor_type is None: 113 pattern = "*.h5" 114 elif tumor_type in TUMOR_TYPES.values(): 115 pattern = f"*_{tumor_type}.h5" 116 else: 117 raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.values())}.") 118 119 sample_paths = natsorted(glob(os.path.join(preprocessed_dir, pattern))) 120 assert len(sample_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'." 121 return sample_paths 122 123 124def get_figshare_brain_tumor_dataset( 125 path: Union[os.PathLike, str], 126 patch_shape: Tuple[int, ...], 127 tumor_type: Optional[str] = None, 128 resize_inputs: bool = False, 129 download: bool = False, 130 **kwargs 131) -> Dataset: 132 """Get the Figshare Brain Tumor dataset for meningioma, glioma and pituitary tumor segmentation. 133 134 Args: 135 path: Filepath to a folder where the data is downloaded for further processing. 136 patch_shape: The patch shape to use for training. 137 tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'. 138 resize_inputs: Whether to resize inputs to the desired patch shape. 139 download: Whether to download the data if it is not present. 140 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 141 142 Returns: 143 The segmentation dataset. 144 """ 145 sample_paths = get_figshare_brain_tumor_paths(path, tumor_type, download) 146 147 if resize_inputs: 148 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 149 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 150 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 151 ) 152 153 return torch_em.default_segmentation_dataset( 154 raw_paths=sample_paths, 155 raw_key="raw", 156 label_paths=sample_paths, 157 label_key="labels", 158 patch_shape=patch_shape, 159 is_seg_dataset=True, 160 **kwargs 161 ) 162 163 164def get_figshare_brain_tumor_loader( 165 path: Union[os.PathLike, str], 166 batch_size: int, 167 patch_shape: Tuple[int, ...], 168 tumor_type: Optional[str] = None, 169 resize_inputs: bool = False, 170 download: bool = False, 171 **kwargs 172) -> DataLoader: 173 """Get the Figshare Brain Tumor dataloader for meningioma, glioma and pituitary tumor segmentation. 174 175 Args: 176 path: Filepath to a folder where the data is downloaded for further processing. 177 batch_size: The batch size for training. 178 patch_shape: The patch shape to use for training. 179 tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'. 180 resize_inputs: Whether to resize inputs to the desired patch shape. 181 download: Whether to download the data if it is not present. 182 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 183 184 Returns: 185 The DataLoader. 186 """ 187 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 188 dataset = get_figshare_brain_tumor_dataset(path, patch_shape, tumor_type, resize_inputs, download, **ds_kwargs) 189 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)