torch_em.data.datasets.medical.curious2022
The CuRIOUS2022 dataset contains annotations for brain tumor segmentation (before resection) and resection cavity segmentation (after resection) in intra-operative brain ultrasound.
This is the segmentation track of the CuRIOUS 2022 MICCAI challenge, located at https://curious2022.grand-challenge.org. The underlying ultrasound (and MRI) volumes are from the RESECT database, located at https://doi.org/10.11582/2017.00004, and the segmentation annotations are from the RESECT-SEG dataset, located at https://osf.io/jv8bk (published under CC-BY-NC-SA-4.0). Please cite the following publications if you use this dataset in your research:
- Y. Xiao et al., "REtroSpective Evaluation of Cerebral Tumors (RESECT): A clinical database of pre-operative MRI and intra-operative ultrasound in low-grade glioma surgeries", Medical Physics, 2017. https://doi.org/10.1002/mp.12268
- B. Behboodi et al., "Open access segmentations of intraoperative brain tumor ultrasound images", Medical Physics, 2024. https://doi.org/10.1002/mp.17317
1"""The CuRIOUS2022 dataset contains annotations for brain tumor segmentation (before resection) and 2resection cavity segmentation (after resection) in intra-operative brain ultrasound. 3 4This is the segmentation track of the CuRIOUS 2022 MICCAI challenge, located at 5https://curious2022.grand-challenge.org. The underlying ultrasound (and MRI) volumes are from the 6RESECT database, located at https://doi.org/10.11582/2017.00004, and the segmentation annotations are 7from the RESECT-SEG dataset, located at https://osf.io/jv8bk (published under CC-BY-NC-SA-4.0). 8Please cite the following publications if you use this dataset in your research: 9- Y. Xiao et al., "REtroSpective Evaluation of Cerebral Tumors (RESECT): A clinical database of 10 pre-operative MRI and intra-operative ultrasound in low-grade glioma surgeries", Medical Physics, 2017. 11 https://doi.org/10.1002/mp.12268 12- B. Behboodi et al., "Open access segmentations of intraoperative brain tumor ultrasound images", 13 Medical Physics, 2024. https://doi.org/10.1002/mp.17317 14""" 15 16import os 17from typing import Union, Tuple, Literal, List 18 19import requests 20 21from torch.utils.data import Dataset, DataLoader 22 23import torch_em 24 25from .. import util 26 27 28RESECT_DATASET_ID = "5686d8fa-2003-4837-8e66-8e887fabe21e" 29RESECT_BASE_URL = f"https://data.archive.sigma2.no/dataset/{RESECT_DATASET_ID}/download/RESECT/NIFTI" 30 31RESECT_SEG_NODE_ID = "jv8bk" 32RESECT_SEG_ROOT_FOLDER_ID = "64cd20819cbf033b051e46c8" 33 34CASE_IDS = [2, 3, 7, 8, 11, 12, 15, 17, 18, 23] 35"""The RESECT case ids for which the RESECT-SEG dataset provides tumor and / or resection cavity 36segmentations. Note that Case11 does not have a resection cavity annotation.""" 37 38 39def _osf_list_all(url): 40 items = [] 41 while url: 42 r = requests.get(url) 43 r.raise_for_status() 44 payload = r.json() 45 items.extend(payload["data"]) 46 url = payload["links"].get("next") 47 return items 48 49 50def _get_seg_case_folder_ids(): 51 root_url = f"https://api.osf.io/v2/nodes/{RESECT_SEG_NODE_ID}/files/osfstorage/{RESECT_SEG_ROOT_FOLDER_ID}/" 52 items = _osf_list_all(root_url) 53 return { 54 item["attributes"]["name"]: item["id"] for item in items if item["attributes"]["kind"] == "folder" 55 } 56 57 58def _get_seg_file_urls(folder_id): 59 url = f"https://api.osf.io/v2/nodes/{RESECT_SEG_NODE_ID}/files/osfstorage/{folder_id}/" 60 items = _osf_list_all(url) 61 return {item["attributes"]["name"]: item["links"]["download"] for item in items} 62 63 64def get_curious2022_data(path: Union[os.PathLike, str], download: bool = False) -> str: 65 """Download the CuRIOUS2022 dataset. 66 67 Args: 68 path: Filepath to a folder where the data is downloaded for further processing. 69 download: Whether to download the data if it is not present. 70 71 Returns: 72 Filepath where the data is downloaded. 73 """ 74 os.makedirs(path, exist_ok=True) 75 76 case_folder_ids = None 77 for case_id in CASE_IDS: 78 case_dir = os.path.join(path, f"Case{case_id}") 79 os.makedirs(case_dir, exist_ok=True) 80 81 for stage in ["before", "after"]: 82 fname = f"Case{case_id}-US-{stage}.nii.gz" 83 dst = os.path.join(case_dir, fname) 84 if os.path.exists(dst): 85 continue 86 url = f"{RESECT_BASE_URL}/Case{case_id}/US/{fname}" 87 util.download_source(path=dst, url=url, download=download) 88 89 for label_type, stage in [("tumor", "before"), ("resection", "after")]: 90 fname = f"Case{case_id}-US-{stage}-{label_type}.nii.gz" 91 dst = os.path.join(case_dir, fname) 92 if os.path.exists(dst): 93 continue 94 if not download: 95 # Case11 does not have a resection cavity annotation, so we cannot know without 96 # querying the OSF API whether the file is expected to exist. 97 continue 98 99 if case_folder_ids is None: 100 case_folder_ids = _get_seg_case_folder_ids() 101 102 folder_id = case_folder_ids.get(f"Case{case_id}") 103 if folder_id is None: 104 continue 105 106 file_urls = _get_seg_file_urls(folder_id) 107 url = file_urls.get(fname) 108 if url is None: # e.g. Case11 has no resection cavity annotation. 109 continue 110 111 util.download_source(path=dst, url=url, download=download) 112 113 return path 114 115 116def get_curious2022_paths( 117 path: Union[os.PathLike, str], task: Literal["tumor", "resection"] = "tumor", download: bool = False, 118) -> Tuple[List[str], List[str]]: 119 """Get paths to the CuRIOUS2022 data. 120 121 Args: 122 path: Filepath to a folder where the data is downloaded for further processing. 123 task: The choice of segmentation task. Either 'tumor' (before resection) or 124 'resection' (resection cavity, after resection). 125 download: Whether to download the data if it is not present. 126 127 Returns: 128 List of filepaths for the image data. 129 List of filepaths for the label data. 130 """ 131 if task not in ["tumor", "resection"]: 132 raise ValueError(f"'{task}' is not a valid task. Choose either 'tumor' or 'resection'.") 133 134 get_curious2022_data(path, download) 135 136 stage = "before" if task == "tumor" else "after" 137 138 raw_paths, label_paths = [], [] 139 for case_id in CASE_IDS: 140 case_dir = os.path.join(path, f"Case{case_id}") 141 raw_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}.nii.gz") 142 label_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}-{task}.nii.gz") 143 if os.path.exists(raw_path) and os.path.exists(label_path): 144 raw_paths.append(raw_path) 145 label_paths.append(label_path) 146 147 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 148 149 return raw_paths, label_paths 150 151 152def get_curious2022_dataset( 153 path: Union[os.PathLike, str], 154 patch_shape: Tuple[int, ...], 155 task: Literal["tumor", "resection"] = "tumor", 156 resize_inputs: bool = False, 157 download: bool = False, 158 **kwargs 159) -> Dataset: 160 """Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative 161 brain ultrasound. 162 163 Args: 164 path: Filepath to a folder where the data is downloaded for further processing. 165 patch_shape: The patch shape to use for training. 166 task: The choice of segmentation task. Either 'tumor' (before resection) or 167 'resection' (resection cavity, after resection). 168 resize_inputs: Whether to resize the inputs. 169 download: Whether to download the data if it is not present. 170 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 171 172 Returns: 173 The segmentation dataset. 174 """ 175 raw_paths, label_paths = get_curious2022_paths(path, task, download) 176 177 if resize_inputs: 178 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 179 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 180 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 181 ) 182 183 return torch_em.default_segmentation_dataset( 184 raw_paths=raw_paths, 185 raw_key="data", 186 label_paths=label_paths, 187 label_key="data", 188 patch_shape=patch_shape, 189 is_seg_dataset=True, 190 **kwargs 191 ) 192 193 194def get_curious2022_loader( 195 path: Union[os.PathLike, str], 196 batch_size: int, 197 patch_shape: Tuple[int, ...], 198 task: Literal["tumor", "resection"] = "tumor", 199 resize_inputs: bool = False, 200 download: bool = False, 201 **kwargs 202) -> DataLoader: 203 """Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative 204 brain ultrasound. 205 206 Args: 207 path: Filepath to a folder where the data is downloaded for further processing. 208 batch_size: The batch size for training. 209 patch_shape: The patch shape to use for training. 210 task: The choice of segmentation task. Either 'tumor' (before resection) or 211 'resection' (resection cavity, after resection). 212 resize_inputs: Whether to resize the inputs. 213 download: Whether to download the data if it is not present. 214 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 215 PyTorch DataLoader. 216 217 Returns: 218 The DataLoader. 219 """ 220 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 221 dataset = get_curious2022_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs) 222 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
The RESECT case ids for which the RESECT-SEG dataset provides tumor and / or resection cavity segmentations. Note that Case11 does not have a resection cavity annotation.
65def get_curious2022_data(path: Union[os.PathLike, str], download: bool = False) -> str: 66 """Download the CuRIOUS2022 dataset. 67 68 Args: 69 path: Filepath to a folder where the data is downloaded for further processing. 70 download: Whether to download the data if it is not present. 71 72 Returns: 73 Filepath where the data is downloaded. 74 """ 75 os.makedirs(path, exist_ok=True) 76 77 case_folder_ids = None 78 for case_id in CASE_IDS: 79 case_dir = os.path.join(path, f"Case{case_id}") 80 os.makedirs(case_dir, exist_ok=True) 81 82 for stage in ["before", "after"]: 83 fname = f"Case{case_id}-US-{stage}.nii.gz" 84 dst = os.path.join(case_dir, fname) 85 if os.path.exists(dst): 86 continue 87 url = f"{RESECT_BASE_URL}/Case{case_id}/US/{fname}" 88 util.download_source(path=dst, url=url, download=download) 89 90 for label_type, stage in [("tumor", "before"), ("resection", "after")]: 91 fname = f"Case{case_id}-US-{stage}-{label_type}.nii.gz" 92 dst = os.path.join(case_dir, fname) 93 if os.path.exists(dst): 94 continue 95 if not download: 96 # Case11 does not have a resection cavity annotation, so we cannot know without 97 # querying the OSF API whether the file is expected to exist. 98 continue 99 100 if case_folder_ids is None: 101 case_folder_ids = _get_seg_case_folder_ids() 102 103 folder_id = case_folder_ids.get(f"Case{case_id}") 104 if folder_id is None: 105 continue 106 107 file_urls = _get_seg_file_urls(folder_id) 108 url = file_urls.get(fname) 109 if url is None: # e.g. Case11 has no resection cavity annotation. 110 continue 111 112 util.download_source(path=dst, url=url, download=download) 113 114 return path
Download the CuRIOUS2022 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
117def get_curious2022_paths( 118 path: Union[os.PathLike, str], task: Literal["tumor", "resection"] = "tumor", download: bool = False, 119) -> Tuple[List[str], List[str]]: 120 """Get paths to the CuRIOUS2022 data. 121 122 Args: 123 path: Filepath to a folder where the data is downloaded for further processing. 124 task: The choice of segmentation task. Either 'tumor' (before resection) or 125 'resection' (resection cavity, after resection). 126 download: Whether to download the data if it is not present. 127 128 Returns: 129 List of filepaths for the image data. 130 List of filepaths for the label data. 131 """ 132 if task not in ["tumor", "resection"]: 133 raise ValueError(f"'{task}' is not a valid task. Choose either 'tumor' or 'resection'.") 134 135 get_curious2022_data(path, download) 136 137 stage = "before" if task == "tumor" else "after" 138 139 raw_paths, label_paths = [], [] 140 for case_id in CASE_IDS: 141 case_dir = os.path.join(path, f"Case{case_id}") 142 raw_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}.nii.gz") 143 label_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}-{task}.nii.gz") 144 if os.path.exists(raw_path) and os.path.exists(label_path): 145 raw_paths.append(raw_path) 146 label_paths.append(label_path) 147 148 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 149 150 return raw_paths, label_paths
Get paths to the CuRIOUS2022 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
153def get_curious2022_dataset( 154 path: Union[os.PathLike, str], 155 patch_shape: Tuple[int, ...], 156 task: Literal["tumor", "resection"] = "tumor", 157 resize_inputs: bool = False, 158 download: bool = False, 159 **kwargs 160) -> Dataset: 161 """Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative 162 brain ultrasound. 163 164 Args: 165 path: Filepath to a folder where the data is downloaded for further processing. 166 patch_shape: The patch shape to use for training. 167 task: The choice of segmentation task. Either 'tumor' (before resection) or 168 'resection' (resection cavity, after resection). 169 resize_inputs: Whether to resize the inputs. 170 download: Whether to download the data if it is not present. 171 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 172 173 Returns: 174 The segmentation dataset. 175 """ 176 raw_paths, label_paths = get_curious2022_paths(path, task, download) 177 178 if resize_inputs: 179 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 180 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 181 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 182 ) 183 184 return torch_em.default_segmentation_dataset( 185 raw_paths=raw_paths, 186 raw_key="data", 187 label_paths=label_paths, 188 label_key="data", 189 patch_shape=patch_shape, 190 is_seg_dataset=True, 191 **kwargs 192 )
Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative brain ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
195def get_curious2022_loader( 196 path: Union[os.PathLike, str], 197 batch_size: int, 198 patch_shape: Tuple[int, ...], 199 task: Literal["tumor", "resection"] = "tumor", 200 resize_inputs: bool = False, 201 download: bool = False, 202 **kwargs 203) -> DataLoader: 204 """Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative 205 brain ultrasound. 206 207 Args: 208 path: Filepath to a folder where the data is downloaded for further processing. 209 batch_size: The batch size for training. 210 patch_shape: The patch shape to use for training. 211 task: The choice of segmentation task. Either 'tumor' (before resection) or 212 'resection' (resection cavity, after resection). 213 resize_inputs: Whether to resize the inputs. 214 download: Whether to download the data if it is not present. 215 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 216 PyTorch DataLoader. 217 218 Returns: 219 The DataLoader. 220 """ 221 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 222 dataset = get_curious2022_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs) 223 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative brain ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.