torch_em.data.datasets.medical.phlf
PHLF is a dataset for segmentation of the liver, liver tumor, Couinaud liver segments, spleen and psoas muscle in hepatobiliary-phase Gd-EOB-DTPA-enhanced MRI.
The dataset consists of preoperative MRI scans of 220 patients from three academic medical centers who underwent hepatectomy, together with 22,342 expert annotations of the whole liver, liver tumors, the eight Couinaud liver segments, the spleen and the psoas muscle. The annotations enable automated quantification of future liver remnant (FLR) volume for predicting post-hepatectomy liver failure (PHLF).
The dataset is located at https://doi.org/10.5281/zenodo.18622298 (Zenodo, CC BY 4.0). The dataset is from the publication https://doi.org/10.1038/s41597-026-07483-x. Please cite it if you use this dataset for your research.
1"""PHLF is a dataset for segmentation of the liver, liver tumor, Couinaud liver segments, spleen and 2psoas muscle in hepatobiliary-phase Gd-EOB-DTPA-enhanced MRI. 3 4The dataset consists of preoperative MRI scans of 220 patients from three academic medical centers who 5underwent hepatectomy, together with 22,342 expert annotations of the whole liver, liver tumors, the 6eight Couinaud liver segments, the spleen and the psoas muscle. The annotations enable automated 7quantification of future liver remnant (FLR) volume for predicting post-hepatectomy liver failure (PHLF). 8 9The dataset is located at https://doi.org/10.5281/zenodo.18622298 (Zenodo, CC BY 4.0). 10The dataset is from the publication https://doi.org/10.1038/s41597-026-07483-x. 11Please cite it if you use this dataset for your research. 12""" 13 14import os 15from glob import glob 16from natsort import natsorted 17from typing import Union, Tuple, Literal, List 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24 25 26URL = "https://zenodo.org/api/records/18622298/files/PHLF.zip/content" 27 28CENTERS = ("Center1", "Center2", "Center3") 29 30LABEL_FOLDERS = { 31 "liver": "Annotation_Whole liver", 32 "tumor": "Annotation_Liver tumor", 33 "couinaud": "Annotation_Couinaud liver segments", 34 "spleen": "Annotation_Spleen", 35 "psoas": "Annotation_Muscle", 36} 37 38 39def get_phlf_data(path: Union[os.PathLike, str], download: bool = False) -> str: 40 """Download the PHLF data. 41 42 Args: 43 path: Filepath to a folder where the data is downloaded for further processing. 44 download: Whether to download the data if it is not present. 45 46 Returns: 47 Filepath where the data is downloaded. 48 """ 49 data_dir = os.path.join(path, "PHLF") 50 if os.path.exists(data_dir): 51 return data_dir 52 53 os.makedirs(path, exist_ok=True) 54 55 zip_path = os.path.join(path, "PHLF.zip") 56 util.download_source(path=zip_path, url=URL, download=download) 57 util.unzip(zip_path=zip_path, dst=path) 58 59 return data_dir 60 61 62def get_phlf_paths( 63 path: Union[os.PathLike, str], 64 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 65 download: bool = False, 66) -> Tuple[List[str], List[str]]: 67 """Get paths to the PHLF data. 68 69 Args: 70 path: Filepath to a folder where the data is downloaded for further processing. 71 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 72 download: Whether to download the data if it is not present. 73 74 Returns: 75 List of filepaths for the image data. 76 List of filepaths for the label data. 77 """ 78 import nibabel as nib 79 80 data_dir = get_phlf_data(path, download) 81 82 if label_choice not in LABEL_FOLDERS: 83 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(LABEL_FOLDERS.keys())}.") 84 85 raw_paths, label_paths = [], [] 86 for center in CENTERS: 87 label_dir = os.path.join(data_dir, center, LABEL_FOLDERS[label_choice]) 88 for label_path in natsorted(glob(os.path.join(label_dir, "*.nii.gz"))): 89 fname = os.path.basename(label_path) 90 raw_path = os.path.join(data_dir, center, "Image", fname) 91 if not os.path.exists(raw_path): 92 continue 93 # A small number of cases (4 / 220) have an annotation volume with a different shape than 94 # the released image volume, likely because the annotation was made on the original DICOM 95 # resolution rather than the resampled Nifti. These cases are skipped. 96 if nib.load(raw_path).shape != nib.load(label_path).shape: 97 continue 98 raw_paths.append(raw_path) 99 label_paths.append(label_path) 100 101 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 102 103 return raw_paths, label_paths 104 105 106def get_phlf_dataset( 107 path: Union[os.PathLike, str], 108 patch_shape: Tuple[int, ...], 109 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 110 resize_inputs: bool = False, 111 download: bool = False, 112 **kwargs 113) -> Dataset: 114 """Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI. 115 116 Args: 117 path: Filepath to a folder where the data is downloaded for further processing. 118 patch_shape: The patch shape to use for training. 119 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 120 resize_inputs: Whether to resize inputs to the desired patch shape. 121 download: Whether to download the data if it is not present. 122 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 123 124 Returns: 125 The segmentation dataset. 126 """ 127 raw_paths, label_paths = get_phlf_paths(path, label_choice, download) 128 129 if resize_inputs: 130 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 131 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 132 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 133 ) 134 135 return torch_em.default_segmentation_dataset( 136 raw_paths=raw_paths, 137 raw_key="data", 138 label_paths=label_paths, 139 label_key="data", 140 patch_shape=patch_shape, 141 is_seg_dataset=True, 142 **kwargs 143 ) 144 145 146def get_phlf_loader( 147 path: Union[os.PathLike, str], 148 batch_size: int, 149 patch_shape: Tuple[int, ...], 150 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 151 resize_inputs: bool = False, 152 download: bool = False, 153 **kwargs 154) -> DataLoader: 155 """Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI. 156 157 Args: 158 path: Filepath to a folder where the data is downloaded for further processing. 159 batch_size: The batch size for training. 160 patch_shape: The patch shape to use for training. 161 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 162 resize_inputs: Whether to resize inputs to the desired patch shape. 163 download: Whether to download the data if it is not present. 164 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 165 166 Returns: 167 The DataLoader. 168 """ 169 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 170 dataset = get_phlf_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 171 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
40def get_phlf_data(path: Union[os.PathLike, str], download: bool = False) -> str: 41 """Download the PHLF data. 42 43 Args: 44 path: Filepath to a folder where the data is downloaded for further processing. 45 download: Whether to download the data if it is not present. 46 47 Returns: 48 Filepath where the data is downloaded. 49 """ 50 data_dir = os.path.join(path, "PHLF") 51 if os.path.exists(data_dir): 52 return data_dir 53 54 os.makedirs(path, exist_ok=True) 55 56 zip_path = os.path.join(path, "PHLF.zip") 57 util.download_source(path=zip_path, url=URL, download=download) 58 util.unzip(zip_path=zip_path, dst=path) 59 60 return data_dir
Download the PHLF data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
63def get_phlf_paths( 64 path: Union[os.PathLike, str], 65 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 66 download: bool = False, 67) -> Tuple[List[str], List[str]]: 68 """Get paths to the PHLF data. 69 70 Args: 71 path: Filepath to a folder where the data is downloaded for further processing. 72 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 73 download: Whether to download the data if it is not present. 74 75 Returns: 76 List of filepaths for the image data. 77 List of filepaths for the label data. 78 """ 79 import nibabel as nib 80 81 data_dir = get_phlf_data(path, download) 82 83 if label_choice not in LABEL_FOLDERS: 84 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(LABEL_FOLDERS.keys())}.") 85 86 raw_paths, label_paths = [], [] 87 for center in CENTERS: 88 label_dir = os.path.join(data_dir, center, LABEL_FOLDERS[label_choice]) 89 for label_path in natsorted(glob(os.path.join(label_dir, "*.nii.gz"))): 90 fname = os.path.basename(label_path) 91 raw_path = os.path.join(data_dir, center, "Image", fname) 92 if not os.path.exists(raw_path): 93 continue 94 # A small number of cases (4 / 220) have an annotation volume with a different shape than 95 # the released image volume, likely because the annotation was made on the original DICOM 96 # resolution rather than the resampled Nifti. These cases are skipped. 97 if nib.load(raw_path).shape != nib.load(label_path).shape: 98 continue 99 raw_paths.append(raw_path) 100 label_paths.append(label_path) 101 102 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 103 104 return raw_paths, label_paths
Get paths to the PHLF data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
107def get_phlf_dataset( 108 path: Union[os.PathLike, str], 109 patch_shape: Tuple[int, ...], 110 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 111 resize_inputs: bool = False, 112 download: bool = False, 113 **kwargs 114) -> Dataset: 115 """Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI. 116 117 Args: 118 path: Filepath to a folder where the data is downloaded for further processing. 119 patch_shape: The patch shape to use for training. 120 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 121 resize_inputs: Whether to resize inputs to the desired patch shape. 122 download: Whether to download the data if it is not present. 123 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 124 125 Returns: 126 The segmentation dataset. 127 """ 128 raw_paths, label_paths = get_phlf_paths(path, label_choice, download) 129 130 if resize_inputs: 131 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 132 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 133 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 134 ) 135 136 return torch_em.default_segmentation_dataset( 137 raw_paths=raw_paths, 138 raw_key="data", 139 label_paths=label_paths, 140 label_key="data", 141 patch_shape=patch_shape, 142 is_seg_dataset=True, 143 **kwargs 144 )
Get the PHLF dataset for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
147def get_phlf_loader( 148 path: Union[os.PathLike, str], 149 batch_size: int, 150 patch_shape: Tuple[int, ...], 151 label_choice: Literal["liver", "tumor", "couinaud", "spleen", "psoas"] = "liver", 152 resize_inputs: bool = False, 153 download: bool = False, 154 **kwargs 155) -> DataLoader: 156 """Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI. 157 158 Args: 159 path: Filepath to a folder where the data is downloaded for further processing. 160 batch_size: The batch size for training. 161 patch_shape: The patch shape to use for training. 162 label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'. 163 resize_inputs: Whether to resize inputs to the desired patch shape. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 166 167 Returns: 168 The DataLoader. 169 """ 170 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 171 dataset = get_phlf_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 172 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the PHLF dataloader for liver, tumor, Couinaud segment, spleen and psoas segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of annotated structure. One of 'liver', 'tumor', 'couinaud', 'spleen', 'psoas'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.