torch_em.data.datasets.medical.openswisshcc

The OpenSwissHCC dataset contains annotations for liver and hepatocellular carcinoma (HCC) lesion segmentation in multiparametric, multiphasic liver MRI.

The dataset is located at https://doi.org/10.5281/zenodo.21992461. It comprises 132 DCE-MRI examinations (63 HCC-positive, 69 HCC-negative patients) with up to five manually annotated lesions per subject (140 lesions in total, 97 confirmed HCC), each delineated on every sequence and phase where visible. A subset of 16 subjects additionally carries manual whole-liver masks.

The dataset also ships automatically-generated (nnU-Net) whole-liver pseudo-labels for all subjects; these are NOT exposed by this module, which only provides the manual lesion and manual liver annotations.

The dataset is from the publication https://doi.org/10.3390/tomography12090133. Please cite it if you use this dataset for your research.

  1"""The OpenSwissHCC dataset contains annotations for liver and hepatocellular carcinoma (HCC)
  2lesion segmentation in multiparametric, multiphasic liver MRI.
  3
  4The dataset is located at https://doi.org/10.5281/zenodo.21992461. It comprises 132 DCE-MRI
  5examinations (63 HCC-positive, 69 HCC-negative patients) with up to five manually annotated
  6lesions per subject (140 lesions in total, 97 confirmed HCC), each delineated on every sequence
  7and phase where visible. A subset of 16 subjects additionally carries manual whole-liver masks.
  8
  9The dataset also ships automatically-generated (nnU-Net) whole-liver pseudo-labels for all
 10subjects; these are NOT exposed by this module, which only provides the manual lesion and manual
 11liver annotations.
 12
 13The dataset is from the publication https://doi.org/10.3390/tomography12090133.
 14Please cite it if you use this dataset for your research.
 15"""
 16
 17import os
 18import re
 19from glob import glob
 20from pathlib import Path
 21from natsort import natsorted
 22from typing import Union, Tuple, Literal, List
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URLS = {
 32    "sub-001-sub-044": "https://zenodo.org/records/21992461/files/sub-001-sub-044.zip",
 33    "sub-044-sub-088": "https://zenodo.org/records/21992461/files/sub-044-sub-088.zip",
 34    "sub-088-sub-132": "https://zenodo.org/records/21992461/files/sub-088-sub-132.zip",
 35    "derivatives": "https://zenodo.org/records/21992461/files/derivatives.zip",
 36}
 37
 38CHECKSUMS = {
 39    "derivatives": "6734f3b9b484c4d04f2f3258a56ee8552f56894ac25f77fe47f72e85ed0c0308",
 40}
 41"""NOTE: The three large raw MRI archives ('sub-001-sub-044', 'sub-044-sub-088', 'sub-088-sub-132')
 42are not checksummed here (`get_openswisshcc_data` passes `checksum=None` for them), since their
 43size (4-5GB each) makes ad-hoc verification impractical; only the small 'derivatives' archive is
 44checksummed."""
 45
 46LESION_SUFFIX_PATTERN = re.compile(r"-L\d+_seg\.nii\.gz$")
 47LIVER_SUFFIX_PATTERN = re.compile(r"-liver_seg(-annotator)?\.nii\.gz$")
 48
 49
 50def get_openswisshcc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 51    """Download the OpenSwissHCC dataset.
 52
 53    Args:
 54        path: Filepath to a folder where the data is downloaded for further processing.
 55        download: Whether to download the data if it is not present.
 56
 57    Returns:
 58        Filepath where the data is downloaded.
 59    """
 60    raw_dir = os.path.join(path, "raw")
 61    derivatives_dir = os.path.join(path, "derivatives")
 62
 63    os.makedirs(path, exist_ok=True)
 64
 65    for name in ("sub-001-sub-044", "sub-044-sub-088", "sub-088-sub-132"):
 66        if os.path.exists(os.path.join(raw_dir, name)):
 67            continue
 68        zip_path = os.path.join(path, f"{name}.zip")
 69        util.download_source(path=zip_path, url=URLS[name], download=download, checksum=None)
 70        util.unzip(zip_path=zip_path, dst=raw_dir, remove=True)
 71
 72    if not os.path.exists(derivatives_dir):
 73        zip_path = os.path.join(path, "derivatives.zip")
 74        util.download_source(
 75            path=zip_path, url=URLS["derivatives"], download=download, checksum=CHECKSUMS["derivatives"]
 76        )
 77        util.unzip(zip_path=zip_path, dst=path, remove=True)
 78
 79    return path
 80
 81
 82def _build_raw_index(raw_dir):
 83    index = {}
 84    for p in glob(os.path.join(raw_dir, "*", "sub-*", "*", "*.nii.gz")):
 85        parts = Path(p).parts
 86        rel_key = os.path.join(*parts[-3:])  # sub-XXX/<subdir>/<basename>.nii.gz
 87        index[rel_key] = p
 88    return index
 89
 90
 91def get_openswisshcc_paths(
 92    path: Union[os.PathLike, str],
 93    label_choice: Literal["lesion", "liver"] = "lesion",
 94    download: bool = False,
 95) -> Tuple[List[str], List[str]]:
 96    """Get paths to the OpenSwissHCC data.
 97
 98    Args:
 99        path: Filepath to a folder where the data is downloaded for further processing.
100        label_choice: The choice of manual annotation. Either 'lesion' (up to five manually
101            annotated lesions per subject, across all subjects and sequences/phases where the
102            lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of
103            16 subjects).
104        download: Whether to download the data if it is not present.
105
106    Returns:
107        List of filepaths for the image data.
108        List of filepaths for the label data.
109    """
110    if label_choice not in ("lesion", "liver"):
111        raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'lesion' or 'liver'.")
112
113    data_dir = get_openswisshcc_data(path, download)
114    raw_dir = os.path.join(data_dir, "raw")
115    raw_index = _build_raw_index(raw_dir)
116
117    if label_choice == "lesion":
118        derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_lesion_annotations")
119        suffix_pattern = LESION_SUFFIX_PATTERN
120    else:
121        derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_liver_annotations")
122        suffix_pattern = LIVER_SUFFIX_PATTERN
123
124    mask_paths = natsorted(glob(os.path.join(derivatives_subdir, "sub-*", "*", "*.nii.gz")))
125
126    image_paths, gt_paths = [], []
127    for mask_path in mask_paths:
128        parts = Path(mask_path).parts
129        subject, subdir, basename = parts[-3], parts[-2], parts[-1]
130
131        raw_basename = suffix_pattern.sub(".nii.gz", basename)
132        if raw_basename == basename:  # the suffix pattern did not match, skip this file.
133            continue
134
135        rel_key = os.path.join(subject, subdir, raw_basename)
136        raw_path = raw_index.get(rel_key)
137        if raw_path is None:
138            continue
139
140        # A handful of masks in the source archive were annotated on a cropped sub-volume and so
141        # do not share the raw volume's number of slices; skip pairs with a shape mismatch rather
142        # than letting them fail deep in `SegmentationDataset.__init__`.
143        import nibabel as nib
144        if nib.load(raw_path).shape != nib.load(mask_path).shape:
145            continue
146
147        image_paths.append(raw_path)
148        gt_paths.append(mask_path)
149
150    return image_paths, gt_paths
151
152
153def get_openswisshcc_dataset(
154    path: Union[os.PathLike, str],
155    patch_shape: Tuple[int, ...],
156    label_choice: Literal["lesion", "liver"] = "lesion",
157    resize_inputs: bool = False,
158    download: bool = False,
159    **kwargs
160) -> Dataset:
161    """Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI.
162
163    Args:
164        path: Filepath to a folder where the data is downloaded for further processing.
165        patch_shape: The patch shape to use for training.
166        label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
167        resize_inputs: Whether to resize the inputs to the patch shape.
168        download: Whether to download the data if it is not present.
169        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
170
171    Returns:
172        The segmentation dataset.
173    """
174    image_paths, gt_paths = get_openswisshcc_paths(path, label_choice, download)
175
176    if resize_inputs:
177        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
178        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
179            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
180        )
181
182    return torch_em.default_segmentation_dataset(
183        raw_paths=image_paths,
184        raw_key="data",
185        label_paths=gt_paths,
186        label_key="data",
187        patch_shape=patch_shape,
188        is_seg_dataset=True,
189        **kwargs
190    )
191
192
193def get_openswisshcc_loader(
194    path: Union[os.PathLike, str],
195    batch_size: int,
196    patch_shape: Tuple[int, ...],
197    label_choice: Literal["lesion", "liver"] = "lesion",
198    resize_inputs: bool = False,
199    download: bool = False,
200    **kwargs
201) -> DataLoader:
202    """Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI.
203
204    Args:
205        path: Filepath to a folder where the data is downloaded for further processing.
206        batch_size: The batch size for training.
207        patch_shape: The patch shape to use for training.
208        label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
209        resize_inputs: Whether to resize the inputs to the patch shape.
210        download: Whether to download the data if it is not present.
211        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
212
213    Returns:
214        The DataLoader.
215    """
216    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
217    dataset = get_openswisshcc_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
218    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'sub-001-sub-044': 'https://zenodo.org/records/21992461/files/sub-001-sub-044.zip', 'sub-044-sub-088': 'https://zenodo.org/records/21992461/files/sub-044-sub-088.zip', 'sub-088-sub-132': 'https://zenodo.org/records/21992461/files/sub-088-sub-132.zip', 'derivatives': 'https://zenodo.org/records/21992461/files/derivatives.zip'}
CHECKSUMS = {'derivatives': '6734f3b9b484c4d04f2f3258a56ee8552f56894ac25f77fe47f72e85ed0c0308'}

NOTE: The three large raw MRI archives ('sub-001-sub-044', 'sub-044-sub-088', 'sub-088-sub-132') are not checksummed here (get_openswisshcc_data passes checksum=None for them), since their size (4-5GB each) makes ad-hoc verification impractical; only the small 'derivatives' archive is checksummed.

LESION_SUFFIX_PATTERN = re.compile('-L\\d+_seg\\.nii\\.gz$')
LIVER_SUFFIX_PATTERN = re.compile('-liver_seg(-annotator)?\\.nii\\.gz$')
def get_openswisshcc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
51def get_openswisshcc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
52    """Download the OpenSwissHCC dataset.
53
54    Args:
55        path: Filepath to a folder where the data is downloaded for further processing.
56        download: Whether to download the data if it is not present.
57
58    Returns:
59        Filepath where the data is downloaded.
60    """
61    raw_dir = os.path.join(path, "raw")
62    derivatives_dir = os.path.join(path, "derivatives")
63
64    os.makedirs(path, exist_ok=True)
65
66    for name in ("sub-001-sub-044", "sub-044-sub-088", "sub-088-sub-132"):
67        if os.path.exists(os.path.join(raw_dir, name)):
68            continue
69        zip_path = os.path.join(path, f"{name}.zip")
70        util.download_source(path=zip_path, url=URLS[name], download=download, checksum=None)
71        util.unzip(zip_path=zip_path, dst=raw_dir, remove=True)
72
73    if not os.path.exists(derivatives_dir):
74        zip_path = os.path.join(path, "derivatives.zip")
75        util.download_source(
76            path=zip_path, url=URLS["derivatives"], download=download, checksum=CHECKSUMS["derivatives"]
77        )
78        util.unzip(zip_path=zip_path, dst=path, remove=True)
79
80    return path

Download the OpenSwissHCC dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_openswisshcc_paths( path: Union[os.PathLike, str], label_choice: Literal['lesion', 'liver'] = 'lesion', download: bool = False) -> Tuple[List[str], List[str]]:
 92def get_openswisshcc_paths(
 93    path: Union[os.PathLike, str],
 94    label_choice: Literal["lesion", "liver"] = "lesion",
 95    download: bool = False,
 96) -> Tuple[List[str], List[str]]:
 97    """Get paths to the OpenSwissHCC data.
 98
 99    Args:
100        path: Filepath to a folder where the data is downloaded for further processing.
101        label_choice: The choice of manual annotation. Either 'lesion' (up to five manually
102            annotated lesions per subject, across all subjects and sequences/phases where the
103            lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of
104            16 subjects).
105        download: Whether to download the data if it is not present.
106
107    Returns:
108        List of filepaths for the image data.
109        List of filepaths for the label data.
110    """
111    if label_choice not in ("lesion", "liver"):
112        raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'lesion' or 'liver'.")
113
114    data_dir = get_openswisshcc_data(path, download)
115    raw_dir = os.path.join(data_dir, "raw")
116    raw_index = _build_raw_index(raw_dir)
117
118    if label_choice == "lesion":
119        derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_lesion_annotations")
120        suffix_pattern = LESION_SUFFIX_PATTERN
121    else:
122        derivatives_subdir = os.path.join(data_dir, "derivatives", "manual_liver_annotations")
123        suffix_pattern = LIVER_SUFFIX_PATTERN
124
125    mask_paths = natsorted(glob(os.path.join(derivatives_subdir, "sub-*", "*", "*.nii.gz")))
126
127    image_paths, gt_paths = [], []
128    for mask_path in mask_paths:
129        parts = Path(mask_path).parts
130        subject, subdir, basename = parts[-3], parts[-2], parts[-1]
131
132        raw_basename = suffix_pattern.sub(".nii.gz", basename)
133        if raw_basename == basename:  # the suffix pattern did not match, skip this file.
134            continue
135
136        rel_key = os.path.join(subject, subdir, raw_basename)
137        raw_path = raw_index.get(rel_key)
138        if raw_path is None:
139            continue
140
141        # A handful of masks in the source archive were annotated on a cropped sub-volume and so
142        # do not share the raw volume's number of slices; skip pairs with a shape mismatch rather
143        # than letting them fail deep in `SegmentationDataset.__init__`.
144        import nibabel as nib
145        if nib.load(raw_path).shape != nib.load(mask_path).shape:
146            continue
147
148        image_paths.append(raw_path)
149        gt_paths.append(mask_path)
150
151    return image_paths, gt_paths

Get paths to the OpenSwissHCC data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • label_choice: The choice of manual annotation. Either 'lesion' (up to five manually annotated lesions per subject, across all subjects and sequences/phases where the lesion is visible) or 'liver' (manual whole-liver masks, available for a subset of 16 subjects).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_openswisshcc_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], label_choice: Literal['lesion', 'liver'] = 'lesion', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
154def get_openswisshcc_dataset(
155    path: Union[os.PathLike, str],
156    patch_shape: Tuple[int, ...],
157    label_choice: Literal["lesion", "liver"] = "lesion",
158    resize_inputs: bool = False,
159    download: bool = False,
160    **kwargs
161) -> Dataset:
162    """Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI.
163
164    Args:
165        path: Filepath to a folder where the data is downloaded for further processing.
166        patch_shape: The patch shape to use for training.
167        label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
168        resize_inputs: Whether to resize the inputs to the patch shape.
169        download: Whether to download the data if it is not present.
170        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
171
172    Returns:
173        The segmentation dataset.
174    """
175    image_paths, gt_paths = get_openswisshcc_paths(path, label_choice, download)
176
177    if resize_inputs:
178        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
179        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
180            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
181        )
182
183    return torch_em.default_segmentation_dataset(
184        raw_paths=image_paths,
185        raw_key="data",
186        label_paths=gt_paths,
187        label_key="data",
188        patch_shape=patch_shape,
189        is_seg_dataset=True,
190        **kwargs
191    )

Get the OpenSwissHCC dataset for segmentation of liver / HCC lesions in multiphasic MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_openswisshcc_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], label_choice: Literal['lesion', 'liver'] = 'lesion', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
194def get_openswisshcc_loader(
195    path: Union[os.PathLike, str],
196    batch_size: int,
197    patch_shape: Tuple[int, ...],
198    label_choice: Literal["lesion", "liver"] = "lesion",
199    resize_inputs: bool = False,
200    download: bool = False,
201    **kwargs
202) -> DataLoader:
203    """Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI.
204
205    Args:
206        path: Filepath to a folder where the data is downloaded for further processing.
207        batch_size: The batch size for training.
208        patch_shape: The patch shape to use for training.
209        label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
210        resize_inputs: Whether to resize the inputs to the patch shape.
211        download: Whether to download the data if it is not present.
212        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
213
214    Returns:
215        The DataLoader.
216    """
217    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
218    dataset = get_openswisshcc_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
219    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the OpenSwissHCC dataloader for segmentation of liver / HCC lesions in multiphasic MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of manual annotation. Either 'lesion' or 'liver'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.