torch_em.data.datasets.medical.hubmap_hpa

The HuBMAP + HPA dataset contains annotations for functional tissue unit (FTU) segmentation in histopathology images of five human organs: kidney, large intestine, spleen, lung and prostate.

This is the dataset for the "HuBMAP + HPA - Hacking the Human Body" Kaggle competition, located at https://www.kaggle.com/competitions/hubmap-organ-segmentation. It combines tissue images from the Human BioMolecular Atlas Program (HuBMAP) and the Human Protein Atlas (HPA), prepared with different staining protocols and imaged at different resolutions.

NOTE: Only the 'train' split has public annotations. The 'test' split labels are held out by Kaggle for competition scoring, so this loader only exposes the labeled training images.

NOTE: Downloading this dataset requires a Kaggle account that has accepted the competition rules at https://www.kaggle.com/competitions/hubmap-organ-segmentation/rules. Without that, the Kaggle API download fails with an HTTP 403 error, even with valid API credentials.

This dataset is from the publication https://doi.org/10.1038/s42003-023-04848-5. Please cite it if you use this dataset in your research.

  1"""The HuBMAP + HPA dataset contains annotations for functional tissue unit (FTU) segmentation
  2in histopathology images of five human organs: kidney, large intestine, spleen, lung and prostate.
  3
  4This is the dataset for the "HuBMAP + HPA - Hacking the Human Body" Kaggle competition, located at
  5https://www.kaggle.com/competitions/hubmap-organ-segmentation. It combines tissue images from the
  6Human BioMolecular Atlas Program (HuBMAP) and the Human Protein Atlas (HPA), prepared with different
  7staining protocols and imaged at different resolutions.
  8
  9NOTE: Only the 'train' split has public annotations. The 'test' split labels are held out by Kaggle
 10for competition scoring, so this loader only exposes the labeled training images.
 11
 12NOTE: Downloading this dataset requires a Kaggle account that has accepted the competition rules at
 13https://www.kaggle.com/competitions/hubmap-organ-segmentation/rules. Without that, the Kaggle API
 14download fails with an HTTP 403 error, even with valid API credentials.
 15
 16This dataset is from the publication https://doi.org/10.1038/s42003-023-04848-5.
 17Please cite it if you use this dataset in your research.
 18"""
 19
 20import os
 21import csv
 22from natsort import natsorted
 23from typing import List, Optional, Tuple, Union
 24
 25import numpy as np
 26import imageio.v3 as imageio
 27
 28from torch.utils.data import Dataset, DataLoader
 29
 30import torch_em
 31
 32from .. import util
 33
 34# The RLE-encoded masks in 'train.csv' exceed the default field size limit for large images.
 35csv.field_size_limit(10**8)
 36
 37
 38ORGANS = ["kidney", "large_intestine", "spleen", "lung", "prostate"]
 39
 40
 41def _decode_rle(rle: str, shape: Tuple[int, int]) -> np.ndarray:
 42    """Decode a Kaggle-style run-length-encoded mask into a binary array.
 43
 44    The encoding lists 1-indexed (start, length) pairs over a column-major (Fortran order)
 45    flattening of the image, i.e. pixels are numbered from top to bottom, then left to right.
 46    """
 47    height, width = shape
 48    mask = np.zeros(height * width, dtype=np.uint8)
 49
 50    values = [int(v) for v in rle.split()]
 51    starts = values[0::2]
 52    lengths = values[1::2]
 53    for start, length in zip(starts, lengths):
 54        start -= 1
 55        mask[start:start + length] = 1
 56
 57    return mask.reshape((width, height)).T
 58
 59
 60def get_hubmap_hpa_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 61    """Download the HuBMAP + HPA dataset.
 62
 63    Args:
 64        path: Filepath to a folder where the data is downloaded for further processing.
 65        download: Whether to download the data if it is not present.
 66
 67    Returns:
 68        Filepath where the data is downloaded.
 69    """
 70    data_dir = os.path.join(path, "train_images")
 71    if os.path.exists(data_dir):
 72        return path
 73
 74    os.makedirs(path, exist_ok=True)
 75
 76    zip_path = os.path.join(path, "hubmap-organ-segmentation.zip")
 77    util.download_source_kaggle(
 78        path=path, dataset_name="hubmap-organ-segmentation", download=download, competition=True
 79    )
 80    util.unzip(zip_path=zip_path, dst=path)
 81
 82    return path
 83
 84
 85def get_hubmap_hpa_paths(
 86    path: Union[os.PathLike, str],
 87    organ: Optional[Union[str, List[str]]] = None,
 88    download: bool = False,
 89) -> Tuple[List[str], List[str]]:
 90    """Get paths to the HuBMAP + HPA data.
 91
 92    Args:
 93        path: Filepath to a folder where the data is downloaded for further processing.
 94        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
 95        download: Whether to download the data if it is not present.
 96
 97    Returns:
 98        List of filepaths for the image data.
 99        List of filepaths for the label data.
100    """
101    data_dir = get_hubmap_hpa_data(path, download)
102
103    if organ is None:
104        organs = ORGANS
105    else:
106        organs = [organ] if isinstance(organ, str) else organ
107        for _organ in organs:
108            if _organ not in ORGANS:
109                raise ValueError(f"'{_organ}' is not a valid organ. Choose from {ORGANS}.")
110
111    label_dir = os.path.join(data_dir, "masks")
112    os.makedirs(label_dir, exist_ok=True)
113
114    csv_path = os.path.join(data_dir, "train.csv")
115    if not os.path.exists(csv_path):
116        raise RuntimeError(f"Could not find 'train.csv' at {csv_path}.")
117
118    with open(csv_path, "r") as f:
119        rows = list(csv.DictReader(f))
120
121    image_paths, label_paths = [], []
122    for row in rows:
123        organ_name = row["organ"].replace(" ", "_")
124        if organ_name not in organs:
125            continue
126
127        image_id = row["id"]
128        image_path = os.path.join(data_dir, "train_images", f"{image_id}.tiff")
129        if not os.path.exists(image_path):
130            continue
131
132        label_path = os.path.join(label_dir, f"{image_id}.tif")
133        if not os.path.exists(label_path):
134            shape = (int(row["img_height"]), int(row["img_width"]))
135            mask = _decode_rle(row["rle"], shape)
136            imageio.imwrite(label_path, mask, compression="zlib")
137
138        image_paths.append(image_path)
139        label_paths.append(label_path)
140
141    return natsorted(image_paths), natsorted(label_paths)
142
143
144def get_hubmap_hpa_dataset(
145    path: Union[os.PathLike, str],
146    patch_shape: Tuple[int, int],
147    organ: Optional[Union[str, List[str]]] = None,
148    resize_inputs: bool = False,
149    download: bool = False,
150    **kwargs
151) -> Dataset:
152    """Get the HuBMAP + HPA dataset for functional tissue unit segmentation.
153
154    Args:
155        path: Filepath to a folder where the data is downloaded for further processing.
156        patch_shape: The patch shape to use for training.
157        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
158        resize_inputs: Whether to resize inputs to the desired patch shape.
159        download: Whether to download the data if it is not present.
160        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
161
162    Returns:
163        The segmentation dataset.
164    """
165    image_paths, label_paths = get_hubmap_hpa_paths(path, organ, download)
166
167    if resize_inputs:
168        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
169        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
170            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
171        )
172
173    return torch_em.default_segmentation_dataset(
174        raw_paths=image_paths,
175        raw_key=None,
176        label_paths=label_paths,
177        label_key=None,
178        is_seg_dataset=False,
179        patch_shape=patch_shape,
180        **kwargs
181    )
182
183
184def get_hubmap_hpa_loader(
185    path: Union[os.PathLike, str],
186    batch_size: int,
187    patch_shape: Tuple[int, int],
188    organ: Optional[Union[str, List[str]]] = None,
189    resize_inputs: bool = False,
190    download: bool = False,
191    **kwargs
192) -> DataLoader:
193    """Get the HuBMAP + HPA dataloader for functional tissue unit segmentation.
194
195    Args:
196        path: Filepath to a folder where the data is downloaded for further processing.
197        batch_size: The batch size for training.
198        patch_shape: The patch shape to use for training.
199        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
200        resize_inputs: Whether to resize inputs to the desired patch shape.
201        download: Whether to download the data if it is not present.
202        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
203
204    Returns:
205        The DataLoader.
206    """
207    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
208    dataset = get_hubmap_hpa_dataset(path, patch_shape, organ, resize_inputs, download, **ds_kwargs)
209    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
ORGANS = ['kidney', 'large_intestine', 'spleen', 'lung', 'prostate']
def get_hubmap_hpa_data(path: Union[os.PathLike, str], download: bool = False) -> str:
61def get_hubmap_hpa_data(path: Union[os.PathLike, str], download: bool = False) -> str:
62    """Download the HuBMAP + HPA dataset.
63
64    Args:
65        path: Filepath to a folder where the data is downloaded for further processing.
66        download: Whether to download the data if it is not present.
67
68    Returns:
69        Filepath where the data is downloaded.
70    """
71    data_dir = os.path.join(path, "train_images")
72    if os.path.exists(data_dir):
73        return path
74
75    os.makedirs(path, exist_ok=True)
76
77    zip_path = os.path.join(path, "hubmap-organ-segmentation.zip")
78    util.download_source_kaggle(
79        path=path, dataset_name="hubmap-organ-segmentation", download=download, competition=True
80    )
81    util.unzip(zip_path=zip_path, dst=path)
82
83    return path

Download the HuBMAP + HPA dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_hubmap_hpa_paths( path: Union[os.PathLike, str], organ: Union[List[str], str, NoneType] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 86def get_hubmap_hpa_paths(
 87    path: Union[os.PathLike, str],
 88    organ: Optional[Union[str, List[str]]] = None,
 89    download: bool = False,
 90) -> Tuple[List[str], List[str]]:
 91    """Get paths to the HuBMAP + HPA data.
 92
 93    Args:
 94        path: Filepath to a folder where the data is downloaded for further processing.
 95        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
 96        download: Whether to download the data if it is not present.
 97
 98    Returns:
 99        List of filepaths for the image data.
100        List of filepaths for the label data.
101    """
102    data_dir = get_hubmap_hpa_data(path, download)
103
104    if organ is None:
105        organs = ORGANS
106    else:
107        organs = [organ] if isinstance(organ, str) else organ
108        for _organ in organs:
109            if _organ not in ORGANS:
110                raise ValueError(f"'{_organ}' is not a valid organ. Choose from {ORGANS}.")
111
112    label_dir = os.path.join(data_dir, "masks")
113    os.makedirs(label_dir, exist_ok=True)
114
115    csv_path = os.path.join(data_dir, "train.csv")
116    if not os.path.exists(csv_path):
117        raise RuntimeError(f"Could not find 'train.csv' at {csv_path}.")
118
119    with open(csv_path, "r") as f:
120        rows = list(csv.DictReader(f))
121
122    image_paths, label_paths = [], []
123    for row in rows:
124        organ_name = row["organ"].replace(" ", "_")
125        if organ_name not in organs:
126            continue
127
128        image_id = row["id"]
129        image_path = os.path.join(data_dir, "train_images", f"{image_id}.tiff")
130        if not os.path.exists(image_path):
131            continue
132
133        label_path = os.path.join(label_dir, f"{image_id}.tif")
134        if not os.path.exists(label_path):
135            shape = (int(row["img_height"]), int(row["img_width"]))
136            mask = _decode_rle(row["rle"], shape)
137            imageio.imwrite(label_path, mask, compression="zlib")
138
139        image_paths.append(image_path)
140        label_paths.append(label_path)
141
142    return natsorted(image_paths), natsorted(label_paths)

Get paths to the HuBMAP + HPA data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • organ: The choice of organ(s) to subselect the data for. Refer to ORGANS for the supported organs.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_hubmap_hpa_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], organ: Union[List[str], str, NoneType] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
145def get_hubmap_hpa_dataset(
146    path: Union[os.PathLike, str],
147    patch_shape: Tuple[int, int],
148    organ: Optional[Union[str, List[str]]] = None,
149    resize_inputs: bool = False,
150    download: bool = False,
151    **kwargs
152) -> Dataset:
153    """Get the HuBMAP + HPA dataset for functional tissue unit segmentation.
154
155    Args:
156        path: Filepath to a folder where the data is downloaded for further processing.
157        patch_shape: The patch shape to use for training.
158        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
159        resize_inputs: Whether to resize inputs to the desired patch shape.
160        download: Whether to download the data if it is not present.
161        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
162
163    Returns:
164        The segmentation dataset.
165    """
166    image_paths, label_paths = get_hubmap_hpa_paths(path, organ, download)
167
168    if resize_inputs:
169        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
170        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
171            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
172        )
173
174    return torch_em.default_segmentation_dataset(
175        raw_paths=image_paths,
176        raw_key=None,
177        label_paths=label_paths,
178        label_key=None,
179        is_seg_dataset=False,
180        patch_shape=patch_shape,
181        **kwargs
182    )

Get the HuBMAP + HPA dataset for functional tissue unit segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • organ: The choice of organ(s) to subselect the data for. Refer to ORGANS for the supported organs.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_hubmap_hpa_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], organ: Union[List[str], str, NoneType] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
185def get_hubmap_hpa_loader(
186    path: Union[os.PathLike, str],
187    batch_size: int,
188    patch_shape: Tuple[int, int],
189    organ: Optional[Union[str, List[str]]] = None,
190    resize_inputs: bool = False,
191    download: bool = False,
192    **kwargs
193) -> DataLoader:
194    """Get the HuBMAP + HPA dataloader for functional tissue unit segmentation.
195
196    Args:
197        path: Filepath to a folder where the data is downloaded for further processing.
198        batch_size: The batch size for training.
199        patch_shape: The patch shape to use for training.
200        organ: The choice of organ(s) to subselect the data for. Refer to `ORGANS` for the supported organs.
201        resize_inputs: Whether to resize inputs to the desired patch shape.
202        download: Whether to download the data if it is not present.
203        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
204
205    Returns:
206        The DataLoader.
207    """
208    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
209    dataset = get_hubmap_hpa_dataset(path, patch_shape, organ, resize_inputs, download, **ds_kwargs)
210    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the HuBMAP + HPA dataloader for functional tissue unit segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • organ: The choice of organ(s) to subselect the data for. Refer to ORGANS for the supported organs.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.