torch_em.data.datasets.medical.phlf

PHLF is a dataset for segmentation of the liver, liver tumor, Couinaud liver segments, spleen and psoas muscle in hepatobiliary-phase Gd-EOB-DTPA-enhanced MRI.

The dataset consists of preoperative MRI scans of 220 patients from three academic medical centers who underwent hepatectomy, together with 22,342 expert annotations of the whole liver, liver tumors, the eight Couinaud liver segments, the spleen and the psoas muscle. The annotations enable automated quantification of future liver remnant (FLR) volume for predicting post-hepatectomy liver failure (PHLF).

The dataset is located at https://doi.org/10.5281/zenodo.18622298 (Zenodo, CC BY 4.0). The dataset is from the publication https://doi.org/10.1038/s41597-026-07483-x. Please cite it if you use this dataset for your research.

  1"""PHLF is a dataset for segmentation of the liver, liver tumor, Couinaud liver segments, spleen and
  2psoas muscle in hepatobiliary-phase Gd-EOB-DTPA-enhanced MRI.
  3
  4The dataset consists of preoperative MRI scans of 220 patients from three academic medical centers who
  5underwent hepatectomy, together with 22,342 expert annotations of the whole liver, liver tumors, the
  6eight Couinaud liver segments, the spleen and the psoas muscle. The annotations enable automated
  7quantification of future liver remnant (FLR) volume for predicting post-hepatectomy liver failure (PHLF).
  8
  9The dataset is located at https://doi.org/10.5281/zenodo.18622298 (Zenodo, CC BY 4.0).
 10The dataset is from the publication https://doi.org/10.1038/s41597-026-07483-x.
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15from glob import glob
 16from natsort import natsorted
 17from typing import Union, Tuple, Literal, List
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26URL = "https://zenodo.org/api/records/18622298/files/PHLF.zip/content"
 27
 28CENTERS = ("Center1", "Center2", "Center3")
 29
 30LABEL_FOLDERS = {
 31    "liver": "Annotation_Whole liver",
 32    "tumor": "Annotation_Liver tumor",
 33    "couinaud": "Annotation_Couinaud liver segments",
 34    "spleen": "Annotation_Spleen",
 35    "psoas": "Annotation_Muscle",
 36}
 37
 38
 39def get_phlf_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 40    """Download the PHLF data.
 41
 42    Args:
 43        path: Filepath to a folder where the data is downloaded for further processing.
 44        download: Whether to download the data if it is not present.
 45
 46    Returns:
 47        Filepath where the data is downloaded.
 48    """
 49    data_dir = os.path.join(path, "PHLF")
 50    if os.path.exists(data_dir):
 51        return data_dir
 52
 53    os.makedirs(path, exist_ok=True)
 54
 55    zip_path = os.path.join(path, "PHLF.zip")
 56    util.download_source(path=zip_path, url=URL, download=download)
 57    util.unzip(zip_path=zip_path, dst=path)
 58
 59    return data_dir
 60
 61
 62def get_phlf_paths(
 63    path: Union[os.PathLike, str],
 64    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
 65    download: bool = False,
 66) -> Tuple[List[str], List[str]]:
 67    """Get paths to the PHLF data.
 68
 69    Args:
 70        path: Filepath to a folder where the data is downloaded for further processing.
 71        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
 72        download: Whether to download the data if it is not present.
 73
 74    Returns:
 75        List of filepaths for the image data.
 76        List of filepaths for the label data.
 77    """
 78    import nibabel as nib
 79
 80    data_dir = get_phlf_data(path, download)
 81
 82    if label_choice not in LABEL_FOLDERS:
 83        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(LABEL_FOLDERS.keys())}.")
 84
 85    raw_paths, label_paths = [], []
 86    for center in CENTERS:
 87        label_dir = os.path.join(data_dir, center, LABEL_FOLDERS[label_choice])
 88        for label_path in natsorted(glob(os.path.join(label_dir, "*.nii.gz"))):
 89            fname = os.path.basename(label_path)
 90            raw_path = os.path.join(data_dir, center, "Image", fname)
 91            if not os.path.exists(raw_path):
 92                continue
 93            # A small number of cases (4 / 220) have an annotation volume with a different shape than
 94            # the released image volume, likely because the annotation was made on the original DICOM
 95            # resolution rather than the resampled Nifti. These cases are skipped.
 96            if nib.load(raw_path).shape != nib.load(label_path).shape:
 97                continue
 98            raw_paths.append(raw_path)
 99            label_paths.append(label_path)
100
101    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
102
103    return raw_paths, label_paths
104
105
106def get_phlf_dataset(
107    path: Union[os.PathLike, str],
108    patch_shape: Tuple[int, ...],
109    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
110    resize_inputs: bool = False,
111    download: bool = False,
112    **kwargs
113) -> Dataset:
114    """Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
115
116    Args:
117        path: Filepath to a folder where the data is downloaded for further processing.
118        patch_shape: The patch shape to use for training.
119        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
120        resize_inputs: Whether to resize inputs to the desired patch shape.
121        download: Whether to download the data if it is not present.
122        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
123
124    Returns:
125        The segmentation dataset.
126    """
127    raw_paths, label_paths = get_phlf_paths(path, label_choice, download)
128
129    if resize_inputs:
130        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
131        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
132            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
133        )
134
135    return torch_em.default_segmentation_dataset(
136        raw_paths=raw_paths,
137        raw_key="data",
138        label_paths=label_paths,
139        label_key="data",
140        patch_shape=patch_shape,
141        is_seg_dataset=True,
142        **kwargs
143    )
144
145
146def get_phlf_loader(
147    path: Union[os.PathLike, str],
148    batch_size: int,
149    patch_shape: Tuple[int, ...],
150    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
151    resize_inputs: bool = False,
152    download: bool = False,
153    **kwargs
154) -> DataLoader:
155    """Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        batch_size: The batch size for training.
160        patch_shape: The patch shape to use for training.
161        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
162        resize_inputs: Whether to resize inputs to the desired patch shape.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
165
166    Returns:
167        The DataLoader.
168    """
169    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
170    dataset = get_phlf_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
171    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/api/records/18622298/files/PHLF.zip/content'
CENTERS = ('Center1', 'Center2', 'Center3')
LABEL_FOLDERS = {'liver': 'Annotation_Whole liver', 'tumor': 'Annotation_Liver tumor', 'couinaud': 'Annotation_Couinaud liver segments', 'spleen': 'Annotation_Spleen', 'psoas': 'Annotation_Muscle'}
def get_phlf_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40def get_phlf_data(path: Union[os.PathLike, str], download: bool = False) -> str:
41    """Download the PHLF data.
42
43    Args:
44        path: Filepath to a folder where the data is downloaded for further processing.
45        download: Whether to download the data if it is not present.
46
47    Returns:
48        Filepath where the data is downloaded.
49    """
50    data_dir = os.path.join(path, "PHLF")
51    if os.path.exists(data_dir):
52        return data_dir
53
54    os.makedirs(path, exist_ok=True)
55
56    zip_path = os.path.join(path, "PHLF.zip")
57    util.download_source(path=zip_path, url=URL, download=download)
58    util.unzip(zip_path=zip_path, dst=path)
59
60    return data_dir

Download the PHLF data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_phlf_paths( path: Union[os.PathLike, str], label_choice: Literal['liver', 'tumor', 'couinaud', 'spleen', 'psoas'] = 'liver', download: bool = False) -> Tuple[List[str], List[str]]:
 63def get_phlf_paths(
 64    path: Union[os.PathLike, str],
 65    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
 66    download: bool = False,
 67) -> Tuple[List[str], List[str]]:
 68    """Get paths to the PHLF data.
 69
 70    Args:
 71        path: Filepath to a folder where the data is downloaded for further processing.
 72        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
 73        download: Whether to download the data if it is not present.
 74
 75    Returns:
 76        List of filepaths for the image data.
 77        List of filepaths for the label data.
 78    """
 79    import nibabel as nib
 80
 81    data_dir = get_phlf_data(path, download)
 82
 83    if label_choice not in LABEL_FOLDERS:
 84        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(LABEL_FOLDERS.keys())}.")
 85
 86    raw_paths, label_paths = [], []
 87    for center in CENTERS:
 88        label_dir = os.path.join(data_dir, center, LABEL_FOLDERS[label_choice])
 89        for label_path in natsorted(glob(os.path.join(label_dir, "*.nii.gz"))):
 90            fname = os.path.basename(label_path)
 91            raw_path = os.path.join(data_dir, center, "Image", fname)
 92            if not os.path.exists(raw_path):
 93                continue
 94            # A small number of cases (4 / 220) have an annotation volume with a different shape than
 95            # the released image volume, likely because the annotation was made on the original DICOM
 96            # resolution rather than the resampled Nifti. These cases are skipped.
 97            if nib.load(raw_path).shape != nib.load(label_path).shape:
 98                continue
 99            raw_paths.append(raw_path)
100            label_paths.append(label_path)
101
102    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
103
104    return raw_paths, label_paths

Get paths to the PHLF data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_phlf_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], label_choice: Literal['liver', 'tumor', 'couinaud', 'spleen', 'psoas'] = 'liver', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
107def get_phlf_dataset(
108    path: Union[os.PathLike, str],
109    patch_shape: Tuple[int, ...],
110    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
111    resize_inputs: bool = False,
112    download: bool = False,
113    **kwargs
114) -> Dataset:
115    """Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
116
117    Args:
118        path: Filepath to a folder where the data is downloaded for further processing.
119        patch_shape: The patch shape to use for training.
120        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
121        resize_inputs: Whether to resize inputs to the desired patch shape.
122        download: Whether to download the data if it is not present.
123        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
124
125    Returns:
126        The segmentation dataset.
127    """
128    raw_paths, label_paths = get_phlf_paths(path, label_choice, download)
129
130    if resize_inputs:
131        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
132        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
133            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
134        )
135
136    return torch_em.default_segmentation_dataset(
137        raw_paths=raw_paths,
138        raw_key="data",
139        label_paths=label_paths,
140        label_key="data",
141        patch_shape=patch_shape,
142        is_seg_dataset=True,
143        **kwargs
144    )

Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_phlf_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], label_choice: Literal['liver', 'tumor', 'couinaud', 'spleen', 'psoas'] = 'liver', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
147def get_phlf_loader(
148    path: Union[os.PathLike, str],
149    batch_size: int,
150    patch_shape: Tuple[int, ...],
151    label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver",
152    resize_inputs: bool = False,
153    download: bool = False,
154    **kwargs
155) -> DataLoader:
156    """Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        batch_size: The batch size for training.
161        patch_shape: The patch shape to use for training.
162        label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
163        resize_inputs: Whether to resize inputs to the desired patch shape.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_phlf_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.