torch_em.data.datasets.medical.curious2022

The CuRIOUS2022 dataset contains annotations for brain tumor segmentation (before resection) and resection cavity segmentation (after resection) in intra-operative brain ultrasound.

This is the segmentation track of the CuRIOUS 2022 MICCAI challenge, located at https://curious2022.grand-challenge.org. The underlying ultrasound (and MRI) volumes are from the RESECT database, located at https://doi.org/10.11582/2017.00004, and the segmentation annotations are from the RESECT-SEG dataset, located at https://osf.io/jv8bk (published under CC-BY-NC-SA-4.0). Please cite the following publications if you use this dataset in your research:

  • Y. Xiao et al., "REtroSpective Evaluation of Cerebral Tumors (RESECT): A clinical database of pre-operative MRI and intra-operative ultrasound in low-grade glioma surgeries", Medical Physics, 2017. https://doi.org/10.1002/mp.12268
  • B. Behboodi et al., "Open access segmentations of intraoperative brain tumor ultrasound images", Medical Physics, 2024. https://doi.org/10.1002/mp.17317
  1"""The CuRIOUS2022 dataset contains annotations for brain tumor segmentation (before resection) and
  2resection cavity segmentation (after resection) in intra-operative brain ultrasound.
  3
  4This is the segmentation track of the CuRIOUS 2022 MICCAI challenge, located at
  5https://curious2022.grand-challenge.org. The underlying ultrasound (and MRI) volumes are from the
  6RESECT database, located at https://doi.org/10.11582/2017.00004, and the segmentation annotations are
  7from the RESECT-SEG dataset, located at https://osf.io/jv8bk (published under CC-BY-NC-SA-4.0).
  8Please cite the following publications if you use this dataset in your research:
  9- Y. Xiao et al., "REtroSpective Evaluation of Cerebral Tumors (RESECT): A clinical database of
 10  pre-operative MRI and intra-operative ultrasound in low-grade glioma surgeries", Medical Physics, 2017.
 11  https://doi.org/10.1002/mp.12268
 12- B. Behboodi et al., "Open access segmentations of intraoperative brain tumor ultrasound images",
 13  Medical Physics, 2024. https://doi.org/10.1002/mp.17317
 14"""
 15
 16import os
 17from typing import Union, Tuple, Literal, List
 18
 19import requests
 20
 21from torch.utils.data import Dataset, DataLoader
 22
 23import torch_em
 24
 25from .. import util
 26
 27
 28RESECT_DATASET_ID = "5686d8fa-2003-4837-8e66-8e887fabe21e"
 29RESECT_BASE_URL = f"https://data.archive.sigma2.no/dataset/{RESECT_DATASET_ID}/download/RESECT/NIFTI"
 30
 31RESECT_SEG_NODE_ID = "jv8bk"
 32RESECT_SEG_ROOT_FOLDER_ID = "64cd20819cbf033b051e46c8"
 33
 34CASE_IDS = [2, 3, 7, 8, 11, 12, 15, 17, 18, 23]
 35"""The RESECT case ids for which the RESECT-SEG dataset provides tumor and / or resection cavity
 36segmentations. Note that Case11 does not have a resection cavity annotation."""
 37
 38
 39def _osf_list_all(url):
 40    items = []
 41    while url:
 42        r = requests.get(url)
 43        r.raise_for_status()
 44        payload = r.json()
 45        items.extend(payload["data"])
 46        url = payload["links"].get("next")
 47    return items
 48
 49
 50def _get_seg_case_folder_ids():
 51    root_url = f"https://api.osf.io/v2/nodes/{RESECT_SEG_NODE_ID}/files/osfstorage/{RESECT_SEG_ROOT_FOLDER_ID}/"
 52    items = _osf_list_all(root_url)
 53    return {
 54        item["attributes"]["name"]: item["id"] for item in items if item["attributes"]["kind"] == "folder"
 55    }
 56
 57
 58def _get_seg_file_urls(folder_id):
 59    url = f"https://api.osf.io/v2/nodes/{RESECT_SEG_NODE_ID}/files/osfstorage/{folder_id}/"
 60    items = _osf_list_all(url)
 61    return {item["attributes"]["name"]: item["links"]["download"] for item in items}
 62
 63
 64def get_curious2022_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 65    """Download the CuRIOUS2022 dataset.
 66
 67    Args:
 68        path: Filepath to a folder where the data is downloaded for further processing.
 69        download: Whether to download the data if it is not present.
 70
 71    Returns:
 72        Filepath where the data is downloaded.
 73    """
 74    os.makedirs(path, exist_ok=True)
 75
 76    case_folder_ids = None
 77    for case_id in CASE_IDS:
 78        case_dir = os.path.join(path, f"Case{case_id}")
 79        os.makedirs(case_dir, exist_ok=True)
 80
 81        for stage in ["before", "after"]:
 82            fname = f"Case{case_id}-US-{stage}.nii.gz"
 83            dst = os.path.join(case_dir, fname)
 84            if os.path.exists(dst):
 85                continue
 86            url = f"{RESECT_BASE_URL}/Case{case_id}/US/{fname}"
 87            util.download_source(path=dst, url=url, download=download)
 88
 89        for label_type, stage in [("tumor", "before"), ("resection", "after")]:
 90            fname = f"Case{case_id}-US-{stage}-{label_type}.nii.gz"
 91            dst = os.path.join(case_dir, fname)
 92            if os.path.exists(dst):
 93                continue
 94            if not download:
 95                # Case11 does not have a resection cavity annotation, so we cannot know without
 96                # querying the OSF API whether the file is expected to exist.
 97                continue
 98
 99            if case_folder_ids is None:
100                case_folder_ids = _get_seg_case_folder_ids()
101
102            folder_id = case_folder_ids.get(f"Case{case_id}")
103            if folder_id is None:
104                continue
105
106            file_urls = _get_seg_file_urls(folder_id)
107            url = file_urls.get(fname)
108            if url is None:  # e.g. Case11 has no resection cavity annotation.
109                continue
110
111            util.download_source(path=dst, url=url, download=download)
112
113    return path
114
115
116def get_curious2022_paths(
117    path: Union[os.PathLike, str], task: Literal["tumor", "resection"] = "tumor", download: bool = False,
118) -> Tuple[List[str], List[str]]:
119    """Get paths to the CuRIOUS2022 data.
120
121    Args:
122        path: Filepath to a folder where the data is downloaded for further processing.
123        task: The choice of segmentation task. Either 'tumor' (before resection) or
124            'resection' (resection cavity, after resection).
125        download: Whether to download the data if it is not present.
126
127    Returns:
128        List of filepaths for the image data.
129        List of filepaths for the label data.
130    """
131    if task not in ["tumor", "resection"]:
132        raise ValueError(f"'{task}' is not a valid task. Choose either 'tumor' or 'resection'.")
133
134    get_curious2022_data(path, download)
135
136    stage = "before" if task == "tumor" else "after"
137
138    raw_paths, label_paths = [], []
139    for case_id in CASE_IDS:
140        case_dir = os.path.join(path, f"Case{case_id}")
141        raw_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}.nii.gz")
142        label_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}-{task}.nii.gz")
143        if os.path.exists(raw_path) and os.path.exists(label_path):
144            raw_paths.append(raw_path)
145            label_paths.append(label_path)
146
147    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
148
149    return raw_paths, label_paths
150
151
152def get_curious2022_dataset(
153    path: Union[os.PathLike, str],
154    patch_shape: Tuple[int, ...],
155    task: Literal["tumor", "resection"] = "tumor",
156    resize_inputs: bool = False,
157    download: bool = False,
158    **kwargs
159) -> Dataset:
160    """Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative
161    brain ultrasound.
162
163    Args:
164        path: Filepath to a folder where the data is downloaded for further processing.
165        patch_shape: The patch shape to use for training.
166        task: The choice of segmentation task. Either 'tumor' (before resection) or
167            'resection' (resection cavity, after resection).
168        resize_inputs: Whether to resize the inputs.
169        download: Whether to download the data if it is not present.
170        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
171
172    Returns:
173        The segmentation dataset.
174    """
175    raw_paths, label_paths = get_curious2022_paths(path, task, download)
176
177    if resize_inputs:
178        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
179        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
180            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
181        )
182
183    return torch_em.default_segmentation_dataset(
184        raw_paths=raw_paths,
185        raw_key="data",
186        label_paths=label_paths,
187        label_key="data",
188        patch_shape=patch_shape,
189        is_seg_dataset=True,
190        **kwargs
191    )
192
193
194def get_curious2022_loader(
195    path: Union[os.PathLike, str],
196    batch_size: int,
197    patch_shape: Tuple[int, ...],
198    task: Literal["tumor", "resection"] = "tumor",
199    resize_inputs: bool = False,
200    download: bool = False,
201    **kwargs
202) -> DataLoader:
203    """Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative
204    brain ultrasound.
205
206    Args:
207        path: Filepath to a folder where the data is downloaded for further processing.
208        batch_size: The batch size for training.
209        patch_shape: The patch shape to use for training.
210        task: The choice of segmentation task. Either 'tumor' (before resection) or
211            'resection' (resection cavity, after resection).
212        resize_inputs: Whether to resize the inputs.
213        download: Whether to download the data if it is not present.
214        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
215            PyTorch DataLoader.
216
217    Returns:
218        The DataLoader.
219    """
220    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
221    dataset = get_curious2022_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs)
222    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
RESECT_DATASET_ID = '5686d8fa-2003-4837-8e66-8e887fabe21e'
RESECT_BASE_URL = 'https://data.archive.sigma2.no/dataset/5686d8fa-2003-4837-8e66-8e887fabe21e/download/RESECT/NIFTI'
RESECT_SEG_NODE_ID = 'jv8bk'
RESECT_SEG_ROOT_FOLDER_ID = '64cd20819cbf033b051e46c8'
CASE_IDS = [2, 3, 7, 8, 11, 12, 15, 17, 18, 23]

The RESECT case ids for which the RESECT-SEG dataset provides tumor and / or resection cavity segmentations. Note that Case11 does not have a resection cavity annotation.

def get_curious2022_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 65def get_curious2022_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 66    """Download the CuRIOUS2022 dataset.
 67
 68    Args:
 69        path: Filepath to a folder where the data is downloaded for further processing.
 70        download: Whether to download the data if it is not present.
 71
 72    Returns:
 73        Filepath where the data is downloaded.
 74    """
 75    os.makedirs(path, exist_ok=True)
 76
 77    case_folder_ids = None
 78    for case_id in CASE_IDS:
 79        case_dir = os.path.join(path, f"Case{case_id}")
 80        os.makedirs(case_dir, exist_ok=True)
 81
 82        for stage in ["before", "after"]:
 83            fname = f"Case{case_id}-US-{stage}.nii.gz"
 84            dst = os.path.join(case_dir, fname)
 85            if os.path.exists(dst):
 86                continue
 87            url = f"{RESECT_BASE_URL}/Case{case_id}/US/{fname}"
 88            util.download_source(path=dst, url=url, download=download)
 89
 90        for label_type, stage in [("tumor", "before"), ("resection", "after")]:
 91            fname = f"Case{case_id}-US-{stage}-{label_type}.nii.gz"
 92            dst = os.path.join(case_dir, fname)
 93            if os.path.exists(dst):
 94                continue
 95            if not download:
 96                # Case11 does not have a resection cavity annotation, so we cannot know without
 97                # querying the OSF API whether the file is expected to exist.
 98                continue
 99
100            if case_folder_ids is None:
101                case_folder_ids = _get_seg_case_folder_ids()
102
103            folder_id = case_folder_ids.get(f"Case{case_id}")
104            if folder_id is None:
105                continue
106
107            file_urls = _get_seg_file_urls(folder_id)
108            url = file_urls.get(fname)
109            if url is None:  # e.g. Case11 has no resection cavity annotation.
110                continue
111
112            util.download_source(path=dst, url=url, download=download)
113
114    return path

Download the CuRIOUS2022 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_curious2022_paths( path: Union[os.PathLike, str], task: Literal['tumor', 'resection'] = 'tumor', download: bool = False) -> Tuple[List[str], List[str]]:
117def get_curious2022_paths(
118    path: Union[os.PathLike, str], task: Literal["tumor", "resection"] = "tumor", download: bool = False,
119) -> Tuple[List[str], List[str]]:
120    """Get paths to the CuRIOUS2022 data.
121
122    Args:
123        path: Filepath to a folder where the data is downloaded for further processing.
124        task: The choice of segmentation task. Either 'tumor' (before resection) or
125            'resection' (resection cavity, after resection).
126        download: Whether to download the data if it is not present.
127
128    Returns:
129        List of filepaths for the image data.
130        List of filepaths for the label data.
131    """
132    if task not in ["tumor", "resection"]:
133        raise ValueError(f"'{task}' is not a valid task. Choose either 'tumor' or 'resection'.")
134
135    get_curious2022_data(path, download)
136
137    stage = "before" if task == "tumor" else "after"
138
139    raw_paths, label_paths = [], []
140    for case_id in CASE_IDS:
141        case_dir = os.path.join(path, f"Case{case_id}")
142        raw_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}.nii.gz")
143        label_path = os.path.join(case_dir, f"Case{case_id}-US-{stage}-{task}.nii.gz")
144        if os.path.exists(raw_path) and os.path.exists(label_path):
145            raw_paths.append(raw_path)
146            label_paths.append(label_path)
147
148    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
149
150    return raw_paths, label_paths

Get paths to the CuRIOUS2022 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_curious2022_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], task: Literal['tumor', 'resection'] = 'tumor', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
153def get_curious2022_dataset(
154    path: Union[os.PathLike, str],
155    patch_shape: Tuple[int, ...],
156    task: Literal["tumor", "resection"] = "tumor",
157    resize_inputs: bool = False,
158    download: bool = False,
159    **kwargs
160) -> Dataset:
161    """Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative
162    brain ultrasound.
163
164    Args:
165        path: Filepath to a folder where the data is downloaded for further processing.
166        patch_shape: The patch shape to use for training.
167        task: The choice of segmentation task. Either 'tumor' (before resection) or
168            'resection' (resection cavity, after resection).
169        resize_inputs: Whether to resize the inputs.
170        download: Whether to download the data if it is not present.
171        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
172
173    Returns:
174        The segmentation dataset.
175    """
176    raw_paths, label_paths = get_curious2022_paths(path, task, download)
177
178    if resize_inputs:
179        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
180        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
181            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
182        )
183
184    return torch_em.default_segmentation_dataset(
185        raw_paths=raw_paths,
186        raw_key="data",
187        label_paths=label_paths,
188        label_key="data",
189        patch_shape=patch_shape,
190        is_seg_dataset=True,
191        **kwargs
192    )

Get the CuRIOUS2022 dataset for brain tumor / resection cavity segmentation in intra-operative brain ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_curious2022_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], task: Literal['tumor', 'resection'] = 'tumor', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
195def get_curious2022_loader(
196    path: Union[os.PathLike, str],
197    batch_size: int,
198    patch_shape: Tuple[int, ...],
199    task: Literal["tumor", "resection"] = "tumor",
200    resize_inputs: bool = False,
201    download: bool = False,
202    **kwargs
203) -> DataLoader:
204    """Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative
205    brain ultrasound.
206
207    Args:
208        path: Filepath to a folder where the data is downloaded for further processing.
209        batch_size: The batch size for training.
210        patch_shape: The patch shape to use for training.
211        task: The choice of segmentation task. Either 'tumor' (before resection) or
212            'resection' (resection cavity, after resection).
213        resize_inputs: Whether to resize the inputs.
214        download: Whether to download the data if it is not present.
215        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
216            PyTorch DataLoader.
217
218    Returns:
219        The DataLoader.
220    """
221    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
222    dataset = get_curious2022_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs)
223    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the CuRIOUS2022 dataloader for brain tumor / resection cavity segmentation in intra-operative brain ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • task: The choice of segmentation task. Either 'tumor' (before resection) or 'resection' (resection cavity, after resection).
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.