torch_em.data.datasets.medical.figshare_brain_tumor

The Figshare Brain Tumor dataset contains annotations for meningioma, glioma and pituitary tumor segmentation in T1-weighted contrast-enhanced brain MRI.

The dataset consists of 3064 slices from 233 patients, acquired at Nanfang Hospital, Guangzhou, China and General Hospital, Tianjin Medical University, China from 2005 to 2010, with an in-plane resolution of 512 x 512 and pixel size of 0.49 x 0.49 mm^2. Each slice comes with a tumor type label (meningioma: 708 slices, glioma: 1426 slices, pituitary tumor: 930 slices) and a binary tumor mask, both distributed as v7.3 matlab '.mat' files (readable with h5py, since scipy.io.loadmat cannot read this format).

The dataset is located at https://doi.org/10.6084/m9.figshare.1512427 and is distributed under the CC BY 4.0 license.

This dataset is from the publications https://doi.org/10.1371/journal.pone.0140381 and https://doi.org/10.1371/journal.pone.0157112. Please cite them if you use this dataset in your research.

  1"""The Figshare Brain Tumor dataset contains annotations for meningioma, glioma and pituitary tumor
  2segmentation in T1-weighted contrast-enhanced brain MRI.
  3
  4The dataset consists of 3064 slices from 233 patients, acquired at Nanfang Hospital, Guangzhou, China
  5and General Hospital, Tianjin Medical University, China from 2005 to 2010, with an in-plane resolution
  6of 512 x 512 and pixel size of 0.49 x 0.49 mm^2. Each slice comes with a tumor type label (meningioma:
  7708 slices, glioma: 1426 slices, pituitary tumor: 930 slices) and a binary tumor mask, both distributed
  8as v7.3 matlab '.mat' files (readable with `h5py`, since `scipy.io.loadmat` cannot read this format).
  9
 10The dataset is located at https://doi.org/10.6084/m9.figshare.1512427 and is distributed under the
 11CC BY 4.0 license.
 12
 13This dataset is from the publications https://doi.org/10.1371/journal.pone.0140381 and
 14https://doi.org/10.1371/journal.pone.0157112. Please cite them if you use this dataset in your research.
 15"""
 16
 17import os
 18import shutil
 19from glob import glob
 20from tqdm import tqdm
 21from natsort import natsorted
 22from typing import Union, Tuple, List, Optional
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URLS = {
 32    "brainTumorDataPublic_1-766.zip": "https://ndownloader.figshare.com/files/3381290",
 33    "brainTumorDataPublic_767-1532.zip": "https://ndownloader.figshare.com/files/3381296",
 34    "brainTumorDataPublic_1533-2298.zip": "https://ndownloader.figshare.com/files/3381293",
 35    "brainTumorDataPublic_2299-3064.zip": "https://ndownloader.figshare.com/files/3381302",
 36}
 37
 38CHECKSUMS = {
 39    "brainTumorDataPublic_1-766.zip": "cdac0a7f4152cb34dd79b13543da96c37a6928ad8cb823e2662150b1a73ade54",
 40    "brainTumorDataPublic_767-1532.zip": "d18e896d14f7af791cdd3d9b9342f9f974742a06132effde5a5219209862b4ce",
 41    "brainTumorDataPublic_1533-2298.zip": "612d9f506c55c54db5f2f38b6d50741ef97a6d7c32b3e1be76e789674a97aefb",
 42    "brainTumorDataPublic_2299-3064.zip": "9a673c55d139133cfbc57097bf155b5ac9f33684ee97f40bf2afba332d416a3a",
 43}
 44
 45TUMOR_TYPES = {1: "meningioma", 2: "glioma", 3: "pituitary"}
 46
 47
 48def _preprocess_figshare_brain_tumor(mat_dir, preprocessed_dir):
 49    import h5py
 50
 51    os.makedirs(preprocessed_dir, exist_ok=True)
 52    mat_paths = natsorted(glob(os.path.join(mat_dir, "*.mat")))
 53    for mat_path in tqdm(mat_paths, desc="Preprocessing inputs"):
 54        sample_id = os.path.splitext(os.path.basename(mat_path))[0]
 55
 56        with h5py.File(mat_path, "r") as f:
 57            cjdata = f["cjdata"]
 58            image = cjdata["image"][:]
 59            mask = cjdata["tumorMask"][:]
 60            label = int(cjdata["label"][()].item())
 61
 62        out_path = os.path.join(preprocessed_dir, f"{sample_id}_{TUMOR_TYPES[label]}.h5")
 63        with h5py.File(out_path, "w") as f:
 64            f.create_dataset("raw", data=image, compression="gzip")
 65            f.create_dataset("labels", data=mask, compression="gzip")
 66
 67    shutil.rmtree(mat_dir)
 68
 69
 70def get_figshare_brain_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 71    """Download the Figshare Brain Tumor data.
 72
 73    Args:
 74        path: Filepath to a folder where the data is downloaded for further processing.
 75        download: Whether to download the data if it is not present.
 76
 77    Returns:
 78        Filepath where the preprocessed data is stored.
 79    """
 80    preprocessed_dir = os.path.join(path, "preprocessed")
 81    if os.path.exists(preprocessed_dir) and glob(os.path.join(preprocessed_dir, "*.h5")):
 82        return preprocessed_dir
 83
 84    os.makedirs(path, exist_ok=True)
 85
 86    mat_dir = os.path.join(path, "mat_files")
 87    os.makedirs(mat_dir, exist_ok=True)
 88    for fname, url in URLS.items():
 89        zip_path = os.path.join(path, fname)
 90        util.download_source(path=zip_path, url=url, download=download, checksum=CHECKSUMS[fname])
 91        util.unzip(zip_path=zip_path, dst=mat_dir)
 92
 93    _preprocess_figshare_brain_tumor(mat_dir, preprocessed_dir)
 94    return preprocessed_dir
 95
 96
 97def get_figshare_brain_tumor_paths(
 98    path: Union[os.PathLike, str], tumor_type: Optional[str] = None, download: bool = False
 99) -> List[str]:
100    """Get paths to the Figshare Brain Tumor data.
101
102    Args:
103        path: Filepath to a folder where the data is downloaded for further processing.
104        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
105        download: Whether to download the data if it is not present.
106
107    Returns:
108        List of filepaths for the stored data.
109    """
110    preprocessed_dir = get_figshare_brain_tumor_data(path, download)
111
112    if tumor_type is None:
113        pattern = "*.h5"
114    elif tumor_type in TUMOR_TYPES.values():
115        pattern = f"*_{tumor_type}.h5"
116    else:
117        raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.values())}.")
118
119    sample_paths = natsorted(glob(os.path.join(preprocessed_dir, pattern)))
120    assert len(sample_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'."
121    return sample_paths
122
123
124def get_figshare_brain_tumor_dataset(
125    path: Union[os.PathLike, str],
126    patch_shape: Tuple[int, ...],
127    tumor_type: Optional[str] = None,
128    resize_inputs: bool = False,
129    download: bool = False,
130    **kwargs
131) -> Dataset:
132    """Get the Figshare Brain Tumor dataset for meningioma, glioma and pituitary tumor segmentation.
133
134    Args:
135        path: Filepath to a folder where the data is downloaded for further processing.
136        patch_shape: The patch shape to use for training.
137        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
138        resize_inputs: Whether to resize inputs to the desired patch shape.
139        download: Whether to download the data if it is not present.
140        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
141
142    Returns:
143        The segmentation dataset.
144    """
145    sample_paths = get_figshare_brain_tumor_paths(path, tumor_type, download)
146
147    if resize_inputs:
148        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
149        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
150            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
151        )
152
153    return torch_em.default_segmentation_dataset(
154        raw_paths=sample_paths,
155        raw_key="raw",
156        label_paths=sample_paths,
157        label_key="labels",
158        patch_shape=patch_shape,
159        is_seg_dataset=True,
160        **kwargs
161    )
162
163
164def get_figshare_brain_tumor_loader(
165    path: Union[os.PathLike, str],
166    batch_size: int,
167    patch_shape: Tuple[int, ...],
168    tumor_type: Optional[str] = None,
169    resize_inputs: bool = False,
170    download: bool = False,
171    **kwargs
172) -> DataLoader:
173    """Get the Figshare Brain Tumor dataloader for meningioma, glioma and pituitary tumor segmentation.
174
175    Args:
176        path: Filepath to a folder where the data is downloaded for further processing.
177        batch_size: The batch size for training.
178        patch_shape: The patch shape to use for training.
179        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
180        resize_inputs: Whether to resize inputs to the desired patch shape.
181        download: Whether to download the data if it is not present.
182        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
183
184    Returns:
185        The DataLoader.
186    """
187    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
188    dataset = get_figshare_brain_tumor_dataset(path, patch_shape, tumor_type, resize_inputs, download, **ds_kwargs)
189    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'brainTumorDataPublic_1-766.zip': 'https://ndownloader.figshare.com/files/3381290', 'brainTumorDataPublic_767-1532.zip': 'https://ndownloader.figshare.com/files/3381296', 'brainTumorDataPublic_1533-2298.zip': 'https://ndownloader.figshare.com/files/3381293', 'brainTumorDataPublic_2299-3064.zip': 'https://ndownloader.figshare.com/files/3381302'}
CHECKSUMS = {'brainTumorDataPublic_1-766.zip': 'cdac0a7f4152cb34dd79b13543da96c37a6928ad8cb823e2662150b1a73ade54', 'brainTumorDataPublic_767-1532.zip': 'd18e896d14f7af791cdd3d9b9342f9f974742a06132effde5a5219209862b4ce', 'brainTumorDataPublic_1533-2298.zip': '612d9f506c55c54db5f2f38b6d50741ef97a6d7c32b3e1be76e789674a97aefb', 'brainTumorDataPublic_2299-3064.zip': '9a673c55d139133cfbc57097bf155b5ac9f33684ee97f40bf2afba332d416a3a'}
TUMOR_TYPES = {1: 'meningioma', 2: 'glioma', 3: 'pituitary'}
def get_figshare_brain_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
71def get_figshare_brain_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
72    """Download the Figshare Brain Tumor data.
73
74    Args:
75        path: Filepath to a folder where the data is downloaded for further processing.
76        download: Whether to download the data if it is not present.
77
78    Returns:
79        Filepath where the preprocessed data is stored.
80    """
81    preprocessed_dir = os.path.join(path, "preprocessed")
82    if os.path.exists(preprocessed_dir) and glob(os.path.join(preprocessed_dir, "*.h5")):
83        return preprocessed_dir
84
85    os.makedirs(path, exist_ok=True)
86
87    mat_dir = os.path.join(path, "mat_files")
88    os.makedirs(mat_dir, exist_ok=True)
89    for fname, url in URLS.items():
90        zip_path = os.path.join(path, fname)
91        util.download_source(path=zip_path, url=url, download=download, checksum=CHECKSUMS[fname])
92        util.unzip(zip_path=zip_path, dst=mat_dir)
93
94    _preprocess_figshare_brain_tumor(mat_dir, preprocessed_dir)
95    return preprocessed_dir

Download the Figshare Brain Tumor data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_figshare_brain_tumor_paths( path: Union[os.PathLike, str], tumor_type: Optional[str] = None, download: bool = False) -> List[str]:
 98def get_figshare_brain_tumor_paths(
 99    path: Union[os.PathLike, str], tumor_type: Optional[str] = None, download: bool = False
100) -> List[str]:
101    """Get paths to the Figshare Brain Tumor data.
102
103    Args:
104        path: Filepath to a folder where the data is downloaded for further processing.
105        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
106        download: Whether to download the data if it is not present.
107
108    Returns:
109        List of filepaths for the stored data.
110    """
111    preprocessed_dir = get_figshare_brain_tumor_data(path, download)
112
113    if tumor_type is None:
114        pattern = "*.h5"
115    elif tumor_type in TUMOR_TYPES.values():
116        pattern = f"*_{tumor_type}.h5"
117    else:
118        raise ValueError(f"'{tumor_type}' is not a valid tumor type. Choose from {list(TUMOR_TYPES.values())}.")
119
120    sample_paths = natsorted(glob(os.path.join(preprocessed_dir, pattern)))
121    assert len(sample_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'."
122    return sample_paths

Get paths to the Figshare Brain Tumor data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the stored data.

def get_figshare_brain_tumor_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], tumor_type: Optional[str] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
125def get_figshare_brain_tumor_dataset(
126    path: Union[os.PathLike, str],
127    patch_shape: Tuple[int, ...],
128    tumor_type: Optional[str] = None,
129    resize_inputs: bool = False,
130    download: bool = False,
131    **kwargs
132) -> Dataset:
133    """Get the Figshare Brain Tumor dataset for meningioma, glioma and pituitary tumor segmentation.
134
135    Args:
136        path: Filepath to a folder where the data is downloaded for further processing.
137        patch_shape: The patch shape to use for training.
138        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
139        resize_inputs: Whether to resize inputs to the desired patch shape.
140        download: Whether to download the data if it is not present.
141        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
142
143    Returns:
144        The segmentation dataset.
145    """
146    sample_paths = get_figshare_brain_tumor_paths(path, tumor_type, download)
147
148    if resize_inputs:
149        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
150        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
151            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
152        )
153
154    return torch_em.default_segmentation_dataset(
155        raw_paths=sample_paths,
156        raw_key="raw",
157        label_paths=sample_paths,
158        label_key="labels",
159        patch_shape=patch_shape,
160        is_seg_dataset=True,
161        **kwargs
162    )

Get the Figshare Brain Tumor dataset for meningioma, glioma and pituitary tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_figshare_brain_tumor_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], tumor_type: Optional[str] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
165def get_figshare_brain_tumor_loader(
166    path: Union[os.PathLike, str],
167    batch_size: int,
168    patch_shape: Tuple[int, ...],
169    tumor_type: Optional[str] = None,
170    resize_inputs: bool = False,
171    download: bool = False,
172    **kwargs
173) -> DataLoader:
174    """Get the Figshare Brain Tumor dataloader for meningioma, glioma and pituitary tumor segmentation.
175
176    Args:
177        path: Filepath to a folder where the data is downloaded for further processing.
178        batch_size: The batch size for training.
179        patch_shape: The patch shape to use for training.
180        tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
181        resize_inputs: Whether to resize inputs to the desired patch shape.
182        download: Whether to download the data if it is not present.
183        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
184
185    Returns:
186        The DataLoader.
187    """
188    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
189    dataset = get_figshare_brain_tumor_dataset(path, patch_shape, tumor_type, resize_inputs, download, **ds_kwargs)
190    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Figshare Brain Tumor dataloader for meningioma, glioma and pituitary tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • tumor_type: The choice of tumor type. One of 'meningioma', 'glioma', 'pituitary'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.