torch_em.data.datasets.medical.openswisshcc
The OpenSwissHCC dataset contains annotations for liver and hepatocellular carcinoma (HCC) lesion segmentation in multiparametric, multiphasic liver MRI.
The dataset is located at https://doi.org/10.5281/zenodo.21992461. It comprises 132 DCE-MRI examinations (63 HCC-positive, 69 HCC-negative patients) with up to five manually annotated lesions per subject (140 lesions in total, 97 confirmed HCC), each delineated on every sequence and phase where visible. A subset of 16 subjects additionally carries manual whole-liver masks.
The dataset also ships automatically-generated (nnU-Net) whole-liver pseudo-labels for all subjects; these are NOT exposed by this module, which only provides the manual lesion and manual liver annotations.
The dataset is from the publication https://doi.org/10.3390/tomography12090133. Please cite it if you use this dataset for your research.
1"""The OpenSwissHCC dataset contains annotations for liver and hepatocellular carcinoma (HCC) 2lesion segmentation in multiparametric, multiphasic liver MRI. 3 4The dataset is located at https://doi.org/10.5281/zenodo.21992461. It comprises 132 DCE-MRI 5examinations (63 HCC-positive, 69 HCC-negative patients) with up to five manually annotated 6lesions per subject (140 lesions in total, 97 confirmed HCC), each delineated on every sequence 7and phase where visible. A subset of 16 subjects additionally carries manual whole-liver masks. 8 9The dataset also ships automatically-generated (nnU-Net) whole-liver pseudo-labels for all 10subjects; these are NOT exposed by this module, which only provides the manual lesion and manual 11liver annotations. 12 13The dataset is from the publication https://doi.org/10.3390/tomography12090133. 14Please cite it if you use this dataset for your research. 15""" 16 17import os 18import re 19from glob import glob 20from pathlib import Path 21from natsort import natsorted 22from typing import Union, Tuple, Literal, List 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31URLS = { 32 "sub-001-sub-044": "https://zenodo.org/records/21992461/files/sub-001-sub-044.zip", 33 "sub-044-sub-088": "https://zenodo.org/records/21992461/files/sub-044-sub-088.zip", 34 "sub-088-sub-132": "https://zenodo.org/records/21992461/files/sub-088-sub-132.zip", 35 "derivatives": "https://zenodo.org/records/21992461/files/derivatives.zip", 36} 37 38CHECKSUMS = { 39 "derivatives": "6734f3b9b484c4d04f2f3258a56ee8552f56894ac25f77fe47f72e85ed0c0308", 40} 41"""NOTE: The three large raw MRI archives ('sub-001-sub-044', 'sub-044-sub-088', 'sub-088-sub-132') 42are not checksummed here (`get_openswisshcc_data` passes `checksum=None` for them), since their 43size (4-5GB each) makes ad-hoc verification impractical; only the small 'derivatives' archive is 44checksummed.""" 45 46LESION_SUFFIX_PATTERN = re.compile(r"-L\d+_seg\.nii\.gz$") 47LIVER_SUFFIX_PATTERN = re.compile(r"-liver_seg(-annotator)?\.nii\.gz$") 48 49 50def get_openswisshcc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 51 """Download the OpenSwissHCC dataset. 52 53 Args: 54 path: Filepath to a folder where the data is downloaded for further processing. 55 download: Whether to download the data if it is not present. 56 57 Returns: 58 Filepath where the data is downloaded. 59 """ 60 raw_dir = os.path.join(path, "raw") 61 derivatives_dir = os.path.join(path, "derivatives") 62 63 os.makedirs(path, exist_ok=True) 64 65 for name in ("sub-001-sub-044", "sub-044-sub-088", "sub-088-sub-132"): 66 if os.path.exists(os.path.join(raw_dir, name)): 67 continue 68 zip_path = os.path.join(path, f"{name}.zip") 69 util.download_source(path=zip_path, url=URLS[name], download=download, checksum=None) 70 util.unzip(zip_path=zip_path, dst=raw_dir, remove=True) 71 72 if not os.path.exists(derivatives_dir): 73 zip_path = os.path.join(path, "derivatives.zip") 74 util.download_source( 75 path=zip_path, url=URLS["derivatives"], download=download, checksum=CHECKSUMS["derivatives"] 76 ) 77 util.unzip(zip_path=zip_path, dst=path, remove=True) 78 79 return path 80 81 82def _build_raw_index(raw_dir): 83 index = {} 84 for p in glob(os.path.join(raw_dir, "*", "sub-*", "*", "*.nii.gz")): 85 parts = Path(p).parts 86 rel_key = os.path.join(*parts[-3:]) # sub-XXX/<subdir>/<basename>.nii.gz 87 index[rel_key] = p 88 return index 89 90 91def get_openswisshcc_paths( 92 path: Union[os.PathLike, str], 93 label_choice: Literal["lesion", "liver"] = "lesion", 94 download: bool = False, 95) -> Tuple[List[str], List[str]]: 96 """Get paths to the OpenSwissHCC data. 97 98 Args: 99 path: Filepath to a folder where the data is downloaded for further processing. 100 label_choice: The choice of manual annotation. Either 'lesion' (up to five manually 101 annotated lesions per subject, across all subjects and sequences/phases where the 102 lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of 103 16 subjects). 104 download: Whether to download the data if it is not present. 105 106 Returns: 107 List of filepaths for the image data. 108 List of filepaths for the label data. 109 """ 110 if label_choice not in ("lesion", "liver"): 111 raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'lesion' or 'liver'.") 112 113 data_dir = get_openswisshcc_data(path, download) 114 raw_dir = os.path.join(data_dir, "raw") 115 raw_index = _build_raw_index(raw_dir) 116 117 if label_choice == "lesion": 118 derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_lesion_annotations") 119 suffix_pattern = LESION_SUFFIX_PATTERN 120 else: 121 derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_liver_annotations") 122 suffix_pattern = LIVER_SUFFIX_PATTERN 123 124 mask_paths = natsorted(glob(os.path.join(derivatives_subdir, "sub-*", "*", "*.nii.gz"))) 125 126 image_paths, gt_paths = [], [] 127 for mask_path in mask_paths: 128 parts = Path(mask_path).parts 129 subject, subdir, basename = parts[-3], parts[-2], parts[-1] 130 131 raw_basename = suffix_pattern.sub(".nii.gz", basename) 132 if raw_basename == basename: # the suffix pattern did not match, skip this file. 133 continue 134 135 rel_key = os.path.join(subject, subdir, raw_basename) 136 raw_path = raw_index.get(rel_key) 137 if raw_path is None: 138 continue 139 140 # A handful of masks in the source archive were annotated on a cropped sub-volume and so 141 # do not share the raw volume's number of slices; skip pairs with a shape mismatch rather 142 # than letting them fail deep in `SegmentationDataset.__init__`. 143 import nibabel as nib 144 if nib.load(raw_path).shape != nib.load(mask_path).shape: 145 continue 146 147 image_paths.append(raw_path) 148 gt_paths.append(mask_path) 149 150 return image_paths, gt_paths 151 152 153def get_openswisshcc_dataset( 154 path: Union[os.PathLike, str], 155 patch_shape: Tuple[int, ...], 156 label_choice: Literal["lesion", "liver"] = "lesion", 157 resize_inputs: bool = False, 158 download: bool = False, 159 **kwargs 160) -> Dataset: 161 """Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI. 162 163 Args: 164 path: Filepath to a folder where the data is downloaded for further processing. 165 patch_shape: The patch shape to use for training. 166 label_choice: The choice of manual annotation. Either 'lesion' or 'liver'. 167 resize_inputs: Whether to resize the inputs to the patch shape. 168 download: Whether to download the data if it is not present. 169 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 170 171 Returns: 172 The segmentation dataset. 173 """ 174 image_paths, gt_paths = get_openswisshcc_paths(path, label_choice, download) 175 176 if resize_inputs: 177 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 178 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 179 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 180 ) 181 182 return torch_em.default_segmentation_dataset( 183 raw_paths=image_paths, 184 raw_key="data", 185 label_paths=gt_paths, 186 label_key="data", 187 patch_shape=patch_shape, 188 is_seg_dataset=True, 189 **kwargs 190 ) 191 192 193def get_openswisshcc_loader( 194 path: Union[os.PathLike, str], 195 batch_size: int, 196 patch_shape: Tuple[int, ...], 197 label_choice: Literal["lesion", "liver"] = "lesion", 198 resize_inputs: bool = False, 199 download: bool = False, 200 **kwargs 201) -> DataLoader: 202 """Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI. 203 204 Args: 205 path: Filepath to a folder where the data is downloaded for further processing. 206 batch_size: The batch size for training. 207 patch_shape: The patch shape to use for training. 208 label_choice: The choice of manual annotation. Either 'lesion' or 'liver'. 209 resize_inputs: Whether to resize the inputs to the patch shape. 210 download: Whether to download the data if it is not present. 211 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 212 213 Returns: 214 The DataLoader. 215 """ 216 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 217 dataset = get_openswisshcc_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 218 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
NOTE: The three large raw MRI archives ('sub-001-sub-044', 'sub-044-sub-088', 'sub-088-sub-132')
are not checksummed here (get_openswisshcc_data passes checksum=None for them), since their
size (4-5GB each) makes ad-hoc verification impractical; only the small 'derivatives' archive is
checksummed.
51def get_openswisshcc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 52 """Download the OpenSwissHCC dataset. 53 54 Args: 55 path: Filepath to a folder where the data is downloaded for further processing. 56 download: Whether to download the data if it is not present. 57 58 Returns: 59 Filepath where the data is downloaded. 60 """ 61 raw_dir = os.path.join(path, "raw") 62 derivatives_dir = os.path.join(path, "derivatives") 63 64 os.makedirs(path, exist_ok=True) 65 66 for name in ("sub-001-sub-044", "sub-044-sub-088", "sub-088-sub-132"): 67 if os.path.exists(os.path.join(raw_dir, name)): 68 continue 69 zip_path = os.path.join(path, f"{name}.zip") 70 util.download_source(path=zip_path, url=URLS[name], download=download, checksum=None) 71 util.unzip(zip_path=zip_path, dst=raw_dir, remove=True) 72 73 if not os.path.exists(derivatives_dir): 74 zip_path = os.path.join(path, "derivatives.zip") 75 util.download_source( 76 path=zip_path, url=URLS["derivatives"], download=download, checksum=CHECKSUMS["derivatives"] 77 ) 78 util.unzip(zip_path=zip_path, dst=path, remove=True) 79 80 return path
Download the OpenSwissHCC dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
92def get_openswisshcc_paths( 93 path: Union[os.PathLike, str], 94 label_choice: Literal["lesion", "liver"] = "lesion", 95 download: bool = False, 96) -> Tuple[List[str], List[str]]: 97 """Get paths to the OpenSwissHCC data. 98 99 Args: 100 path: Filepath to a folder where the data is downloaded for further processing. 101 label_choice: The choice of manual annotation. Either 'lesion' (up to five manually 102 annotated lesions per subject, across all subjects and sequences/phases where the 103 lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of 104 16 subjects). 105 download: Whether to download the data if it is not present. 106 107 Returns: 108 List of filepaths for the image data. 109 List of filepaths for the label data. 110 """ 111 if label_choice not in ("lesion", "liver"): 112 raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'lesion' or 'liver'.") 113 114 data_dir = get_openswisshcc_data(path, download) 115 raw_dir = os.path.join(data_dir, "raw") 116 raw_index = _build_raw_index(raw_dir) 117 118 if label_choice == "lesion": 119 derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_lesion_annotations") 120 suffix_pattern = LESION_SUFFIX_PATTERN 121 else: 122 derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_liver_annotations") 123 suffix_pattern = LIVER_SUFFIX_PATTERN 124 125 mask_paths = natsorted(glob(os.path.join(derivatives_subdir, "sub-*", "*", "*.nii.gz"))) 126 127 image_paths, gt_paths = [], [] 128 for mask_path in mask_paths: 129 parts = Path(mask_path).parts 130 subject, subdir, basename = parts[-3], parts[-2], parts[-1] 131 132 raw_basename = suffix_pattern.sub(".nii.gz", basename) 133 if raw_basename == basename: # the suffix pattern did not match, skip this file. 134 continue 135 136 rel_key = os.path.join(subject, subdir, raw_basename) 137 raw_path = raw_index.get(rel_key) 138 if raw_path is None: 139 continue 140 141 # A handful of masks in the source archive were annotated on a cropped sub-volume and so 142 # do not share the raw volume's number of slices; skip pairs with a shape mismatch rather 143 # than letting them fail deep in `SegmentationDataset.__init__`. 144 import nibabel as nib 145 if nib.load(raw_path).shape != nib.load(mask_path).shape: 146 continue 147 148 image_paths.append(raw_path) 149 gt_paths.append(mask_path) 150 151 return image_paths, gt_paths
Get paths to the OpenSwissHCC data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- label_choice: The choice of manual annotation. Either 'lesion' (up to five manually annotated lesions per subject, across all subjects and sequences/phases where the lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of 16 subjects).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
154def get_openswisshcc_dataset( 155 path: Union[os.PathLike, str], 156 patch_shape: Tuple[int, ...], 157 label_choice: Literal["lesion", "liver"] = "lesion", 158 resize_inputs: bool = False, 159 download: bool = False, 160 **kwargs 161) -> Dataset: 162 """Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI. 163 164 Args: 165 path: Filepath to a folder where the data is downloaded for further processing. 166 patch_shape: The patch shape to use for training. 167 label_choice: The choice of manual annotation. Either 'lesion' or 'liver'. 168 resize_inputs: Whether to resize the inputs to the patch shape. 169 download: Whether to download the data if it is not present. 170 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 171 172 Returns: 173 The segmentation dataset. 174 """ 175 image_paths, gt_paths = get_openswisshcc_paths(path, label_choice, download) 176 177 if resize_inputs: 178 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 179 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 180 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 181 ) 182 183 return torch_em.default_segmentation_dataset( 184 raw_paths=image_paths, 185 raw_key="data", 186 label_paths=gt_paths, 187 label_key="data", 188 patch_shape=patch_shape, 189 is_seg_dataset=True, 190 **kwargs 191 )
Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
194def get_openswisshcc_loader( 195 path: Union[os.PathLike, str], 196 batch_size: int, 197 patch_shape: Tuple[int, ...], 198 label_choice: Literal["lesion", "liver"] = "lesion", 199 resize_inputs: bool = False, 200 download: bool = False, 201 **kwargs 202) -> DataLoader: 203 """Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI. 204 205 Args: 206 path: Filepath to a folder where the data is downloaded for further processing. 207 batch_size: The batch size for training. 208 patch_shape: The patch shape to use for training. 209 label_choice: The choice of manual annotation. Either 'lesion' or 'liver'. 210 resize_inputs: Whether to resize the inputs to the patch shape. 211 download: Whether to download the data if it is not present. 212 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 213 214 Returns: 215 The DataLoader. 216 """ 217 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 218 dataset = get_openswisshcc_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 219 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.