torch_em.data.datasets.medical.hubmap_hpa
The HuBMAP + HPA dataset contains annotations for functional tissue unit (FTU) segmentation in histopathology images of five human organs: kidney, large intestine, spleen, lung and prostate.
This is the dataset for the "HuBMAP + HPA - Hacking the Human Body" Kaggle competition, located at https://www.kaggle.com/competitions/hubmap-organ-segmentation. It combines tissue images from the Human BioMolecular Atlas Program (HuBMAP) and the Human Protein Atlas (HPA), prepared with different staining protocols and imaged at different resolutions.
NOTE: Only the 'train' split has public annotations. The 'test' split labels are held out by Kaggle for competition scoring, so this loader only exposes the labeled training images.
NOTE: Downloading this dataset requires a Kaggle account that has accepted the competition rules at https://www.kaggle.com/competitions/hubmap-organ-segmentation/rules. Without that, the Kaggle API download fails with an HTTP 403 error, even with valid API credentials.
This dataset is from the publication https://doi.org/10.1038/s42003-023-04848-5. Please cite it if you use this dataset in your research.
1"""The HuBMAP + HPA dataset contains annotations for functional tissue unit (FTU) segmentation 2in histopathology images of five human organs: kidney, large intestine, spleen, lung and prostate. 3 4This is the dataset for the "HuBMAP + HPA - Hacking the Human Body" Kaggle competition, located at 5https://www.kaggle.com/competitions/hubmap-organ-segmentation. It combines tissue images from the 6Human BioMolecular Atlas Program (HuBMAP) and the Human Protein Atlas (HPA), prepared with different 7staining protocols and imaged at different resolutions. 8 9NOTE: Only the 'train' split has public annotations. The 'test' split labels are held out by Kaggle 10for competition scoring, so this loader only exposes the labeled training images. 11 12NOTE: Downloading this dataset requires a Kaggle account that has accepted the competition rules at 13https://www.kaggle.com/competitions/hubmap-organ-segmentation/rules. Without that, the Kaggle API 14download fails with an HTTP 403 error, even with valid API credentials. 15 16This dataset is from the publication https://doi.org/10.1038/s42003-023-04848-5. 17Please cite it if you use this dataset in your research. 18""" 19 20import os 21import csv 22from natsort import natsorted 23from typing import List, Optional, Tuple, Union 24 25import numpy as np 26import imageio.v3 as imageio 27 28from torch.utils.data import Dataset, DataLoader 29 30import torch_em 31 32from .. import util 33 34# The RLE-encoded masks in 'train.csv' exceed the default field size limit for large images. 35csv.field_size_limit(10**8) 36 37 38ORGANS = ["kidney", "large_intestine", "spleen", "lung", "prostate"] 39 40 41def _decode_rle(rle: str, shape: Tuple[int, int]) -> np.ndarray: 42 """Decode a Kaggle-style run-length-encoded mask into a binary array. 43 44 The encoding lists 1-indexed (start, length) pairs over a column-major (Fortran order) 45 flattening of the image, i.e. pixels are numbered from top to bottom, then left to right. 46 """ 47 height, width = shape 48 mask = np.zeros(height * width, dtype=np.uint8) 49 50 values = [int(v) for v in rle.split()] 51 starts = values[0::2] 52 lengths = values[1::2] 53 for start, length in zip(starts, lengths): 54 start -= 1 55 mask[start:start + length] = 1 56 57 return mask.reshape((width, height)).T 58 59 60def get_hubmap_hpa_data(path: Union[os.PathLike, str], download: bool = False) -> str: 61 """Download the HuBMAP + HPA dataset. 62 63 Args: 64 path: Filepath to a folder where the data is downloaded for further processing. 65 download: Whether to download the data if it is not present. 66 67 Returns: 68 Filepath where the data is downloaded. 69 """ 70 data_dir = os.path.join(path, "train_images") 71 if os.path.exists(data_dir): 72 return path 73 74 os.makedirs(path, exist_ok=True) 75 76 zip_path = os.path.join(path, "hubmap-organ-segmentation.zip") 77 util.download_source_kaggle( 78 path=path, dataset_name="hubmap-organ-segmentation", download=download, competition=True 79 ) 80 util.unzip(zip_path=zip_path, dst=path) 81 82 return path 83 84 85def get_hubmap_hpa_paths( 86 path: Union[os.PathLike, str], 87 organ: Optional[Union[str, List[str]]] = None, 88 download: bool = False, 89) -> Tuple[List[str], List[str]]: 90 """Get paths to the HuBMAP + HPA data. 91 92 Args: 93 path: Filepath to a folder where the data is downloaded for further processing. 94 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 95 download: Whether to download the data if it is not present. 96 97 Returns: 98 List of filepaths for the image data. 99 List of filepaths for the label data. 100 """ 101 data_dir = get_hubmap_hpa_data(path, download) 102 103 if organ is None: 104 organs = ORGANS 105 else: 106 organs = [organ] if isinstance(organ, str) else organ 107 for _organ in organs: 108 if _organ not in ORGANS: 109 raise ValueError(f"'{_organ}' is not a valid organ. Choose from {ORGANS}.") 110 111 label_dir = os.path.join(data_dir, "masks") 112 os.makedirs(label_dir, exist_ok=True) 113 114 csv_path = os.path.join(data_dir, "train.csv") 115 if not os.path.exists(csv_path): 116 raise RuntimeError(f"Could not find 'train.csv' at {csv_path}.") 117 118 with open(csv_path, "r") as f: 119 rows = list(csv.DictReader(f)) 120 121 image_paths, label_paths = [], [] 122 for row in rows: 123 organ_name = row["organ"].replace(" ", "_") 124 if organ_name not in organs: 125 continue 126 127 image_id = row["id"] 128 image_path = os.path.join(data_dir, "train_images", f"{image_id}.tiff") 129 if not os.path.exists(image_path): 130 continue 131 132 label_path = os.path.join(label_dir, f"{image_id}.tif") 133 if not os.path.exists(label_path): 134 shape = (int(row["img_height"]), int(row["img_width"])) 135 mask = _decode_rle(row["rle"], shape) 136 imageio.imwrite(label_path, mask, compression="zlib") 137 138 image_paths.append(image_path) 139 label_paths.append(label_path) 140 141 return natsorted(image_paths), natsorted(label_paths) 142 143 144def get_hubmap_hpa_dataset( 145 path: Union[os.PathLike, str], 146 patch_shape: Tuple[int, int], 147 organ: Optional[Union[str, List[str]]] = None, 148 resize_inputs: bool = False, 149 download: bool = False, 150 **kwargs 151) -> Dataset: 152 """Get the HuBMAP + HPA dataset for functional tissue unit segmentation. 153 154 Args: 155 path: Filepath to a folder where the data is downloaded for further processing. 156 patch_shape: The patch shape to use for training. 157 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 158 resize_inputs: Whether to resize inputs to the desired patch shape. 159 download: Whether to download the data if it is not present. 160 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 161 162 Returns: 163 The segmentation dataset. 164 """ 165 image_paths, label_paths = get_hubmap_hpa_paths(path, organ, download) 166 167 if resize_inputs: 168 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 169 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 170 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 171 ) 172 173 return torch_em.default_segmentation_dataset( 174 raw_paths=image_paths, 175 raw_key=None, 176 label_paths=label_paths, 177 label_key=None, 178 is_seg_dataset=False, 179 patch_shape=patch_shape, 180 **kwargs 181 ) 182 183 184def get_hubmap_hpa_loader( 185 path: Union[os.PathLike, str], 186 batch_size: int, 187 patch_shape: Tuple[int, int], 188 organ: Optional[Union[str, List[str]]] = None, 189 resize_inputs: bool = False, 190 download: bool = False, 191 **kwargs 192) -> DataLoader: 193 """Get the HuBMAP + HPA dataloader for functional tissue unit segmentation. 194 195 Args: 196 path: Filepath to a folder where the data is downloaded for further processing. 197 batch_size: The batch size for training. 198 patch_shape: The patch shape to use for training. 199 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 200 resize_inputs: Whether to resize inputs to the desired patch shape. 201 download: Whether to download the data if it is not present. 202 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 203 204 Returns: 205 The DataLoader. 206 """ 207 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 208 dataset = get_hubmap_hpa_dataset(path, patch_shape, organ, resize_inputs, download, **ds_kwargs) 209 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
61def get_hubmap_hpa_data(path: Union[os.PathLike, str], download: bool = False) -> str: 62 """Download the HuBMAP + HPA dataset. 63 64 Args: 65 path: Filepath to a folder where the data is downloaded for further processing. 66 download: Whether to download the data if it is not present. 67 68 Returns: 69 Filepath where the data is downloaded. 70 """ 71 data_dir = os.path.join(path, "train_images") 72 if os.path.exists(data_dir): 73 return path 74 75 os.makedirs(path, exist_ok=True) 76 77 zip_path = os.path.join(path, "hubmap-organ-segmentation.zip") 78 util.download_source_kaggle( 79 path=path, dataset_name="hubmap-organ-segmentation", download=download, competition=True 80 ) 81 util.unzip(zip_path=zip_path, dst=path) 82 83 return path
Download the HuBMAP + HPA dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
86def get_hubmap_hpa_paths( 87 path: Union[os.PathLike, str], 88 organ: Optional[Union[str, List[str]]] = None, 89 download: bool = False, 90) -> Tuple[List[str], List[str]]: 91 """Get paths to the HuBMAP + HPA data. 92 93 Args: 94 path: Filepath to a folder where the data is downloaded for further processing. 95 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 96 download: Whether to download the data if it is not present. 97 98 Returns: 99 List of filepaths for the image data. 100 List of filepaths for the label data. 101 """ 102 data_dir = get_hubmap_hpa_data(path, download) 103 104 if organ is None: 105 organs = ORGANS 106 else: 107 organs = [organ] if isinstance(organ, str) else organ 108 for _organ in organs: 109 if _organ not in ORGANS: 110 raise ValueError(f"'{_organ}' is not a valid organ. Choose from {ORGANS}.") 111 112 label_dir = os.path.join(data_dir, "masks") 113 os.makedirs(label_dir, exist_ok=True) 114 115 csv_path = os.path.join(data_dir, "train.csv") 116 if not os.path.exists(csv_path): 117 raise RuntimeError(f"Could not find 'train.csv' at {csv_path}.") 118 119 with open(csv_path, "r") as f: 120 rows = list(csv.DictReader(f)) 121 122 image_paths, label_paths = [], [] 123 for row in rows: 124 organ_name = row["organ"].replace(" ", "_") 125 if organ_name not in organs: 126 continue 127 128 image_id = row["id"] 129 image_path = os.path.join(data_dir, "train_images", f"{image_id}.tiff") 130 if not os.path.exists(image_path): 131 continue 132 133 label_path = os.path.join(label_dir, f"{image_id}.tif") 134 if not os.path.exists(label_path): 135 shape = (int(row["img_height"]), int(row["img_width"])) 136 mask = _decode_rle(row["rle"], shape) 137 imageio.imwrite(label_path, mask, compression="zlib") 138 139 image_paths.append(image_path) 140 label_paths.append(label_path) 141 142 return natsorted(image_paths), natsorted(label_paths)
Get paths to the HuBMAP + HPA data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- organ: The choice of organ(s) to subselect the data for. Refer to
ORGANSfor the supported organs. - download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
145def get_hubmap_hpa_dataset( 146 path: Union[os.PathLike, str], 147 patch_shape: Tuple[int, int], 148 organ: Optional[Union[str, List[str]]] = None, 149 resize_inputs: bool = False, 150 download: bool = False, 151 **kwargs 152) -> Dataset: 153 """Get the HuBMAP + HPA dataset for functional tissue unit segmentation. 154 155 Args: 156 path: Filepath to a folder where the data is downloaded for further processing. 157 patch_shape: The patch shape to use for training. 158 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 159 resize_inputs: Whether to resize inputs to the desired patch shape. 160 download: Whether to download the data if it is not present. 161 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 162 163 Returns: 164 The segmentation dataset. 165 """ 166 image_paths, label_paths = get_hubmap_hpa_paths(path, organ, download) 167 168 if resize_inputs: 169 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 170 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 171 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 172 ) 173 174 return torch_em.default_segmentation_dataset( 175 raw_paths=image_paths, 176 raw_key=None, 177 label_paths=label_paths, 178 label_key=None, 179 is_seg_dataset=False, 180 patch_shape=patch_shape, 181 **kwargs 182 )
Get the HuBMAP + HPA dataset for functional tissue unit segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- organ: The choice of organ(s) to subselect the data for. Refer to
ORGANSfor the supported organs. - resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
185def get_hubmap_hpa_loader( 186 path: Union[os.PathLike, str], 187 batch_size: int, 188 patch_shape: Tuple[int, int], 189 organ: Optional[Union[str, List[str]]] = None, 190 resize_inputs: bool = False, 191 download: bool = False, 192 **kwargs 193) -> DataLoader: 194 """Get the HuBMAP + HPA dataloader for functional tissue unit segmentation. 195 196 Args: 197 path: Filepath to a folder where the data is downloaded for further processing. 198 batch_size: The batch size for training. 199 patch_shape: The patch shape to use for training. 200 organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs. 201 resize_inputs: Whether to resize inputs to the desired patch shape. 202 download: Whether to download the data if it is not present. 203 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 204 205 Returns: 206 The DataLoader. 207 """ 208 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 209 dataset = get_hubmap_hpa_dataset(path, patch_shape, organ, resize_inputs, download, **ds_kwargs) 210 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the HuBMAP + HPA dataloader for functional tissue unit segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- organ: The choice of organ(s) to subselect the data for. Refer to
ORGANSfor the supported organs. - resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.