torch_em.data.datasets.histopathology.imc_kidney

This dataset contains cell instance segmentation annotations for imaging mass cytometry (IMC) of human kidney biopsies with acute interstitial nephritis (AIN), acute tubular injury (ATI), and histologically normal reference tissue.

The data is from the publication "Spatial analysis reveals cellular microenvironments and mechanisms of inflammation and injury in acute interstitial nephritis" and hosted on the BioImage Archive at https://www.ebi.ac.uk/biostudies/bioimages/studies/S-BIAD3694. It is available under the CC0 license. Please cite it if you use this dataset in your research.

This loader covers the Discovery cohort (23 experimental batches, 325 acquisitions), which is provided as processed multichannel TIFF stacks (35-marker antibody panel) with matching per-cell instance segmentation masks generated with Mesmer. The BioImage Archive record also hosts a Validation cohort of raw Hyperion text exports without segmentation masks, which is out of scope for this loader.

  1"""This dataset contains cell instance segmentation annotations for imaging mass cytometry (IMC)
  2of human kidney biopsies with acute interstitial nephritis (AIN), acute tubular injury (ATI), and
  3histologically normal reference tissue.
  4
  5The data is from the publication "Spatial analysis reveals cellular microenvironments and
  6mechanisms of inflammation and injury in acute interstitial nephritis" and hosted on the
  7BioImage Archive at https://www.ebi.ac.uk/biostudies/bioimages/studies/S-BIAD3694. It is
  8available under the CC0 license. Please cite it if you use this dataset in your research.
  9
 10This loader covers the Discovery cohort (23 experimental batches, 325 acquisitions), which is
 11provided as processed multichannel TIFF stacks (35-marker antibody panel) with matching per-cell
 12instance segmentation masks generated with Mesmer. The BioImage Archive record also hosts a
 13Validation cohort of raw Hyperion text exports without segmentation masks, which is out of scope
 14for this loader.
 15"""
 16
 17import os
 18from glob import glob
 19from typing import List, Optional, Sequence, Tuple, Union
 20
 21import pandas as pd
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30BASE_URL = "https://ftp.ebi.ac.uk/pub/databases/biostudies/S-BIAD/694/S-BIAD3694/Files"
 31
 32BATCHES = tuple(f"Batch{i}" for i in range(1, 24))
 33
 34
 35def _get_manifest(path, download):
 36    # NOTE: The per-batch 'images.csv' manifests list acquisitions that were planned but not all of
 37    # them were actually deposited (e.g. excluded ROIs). The top-level file list only contains the
 38    # acquisitions that were actually uploaded, so we rely on it to determine the real image / mask pairs.
 39    manifest_path = os.path.join(path, "bia_filelist_all.tsv")
 40    os.makedirs(path, exist_ok=True)
 41    util.download_source(manifest_path, f"{BASE_URL}/bia_filelist_all.tsv", download, checksum=None)
 42    manifest = pd.read_csv(manifest_path, sep="\t")
 43    return manifest[manifest["cohort"] == "discovery"]
 44
 45
 46def get_imc_kidney_data(
 47    path: Union[os.PathLike, str],
 48    batches: Optional[Sequence[str]] = None,
 49    download: bool = False,
 50) -> str:
 51    """Download the IMC kidney (AIN) Discovery cohort data.
 52
 53    Args:
 54        path: Filepath to a folder where the downloaded data will be saved.
 55        batches: The batch names to prepare. By default all 23 batches are prepared, which
 56            requires downloading several tens of gigabytes of data.
 57        download: Whether to download the data if it is not present.
 58
 59    Returns:
 60        Filepath to the folder where the data is stored.
 61    """
 62    if batches is None:
 63        batches = BATCHES
 64    else:
 65        invalid = sorted(set(batches) - set(BATCHES))
 66        if invalid:
 67            raise ValueError(f"Invalid batch name(s) {invalid}. Choose from {BATCHES}.")
 68
 69    os.makedirs(path, exist_ok=True)
 70    manifest = _get_manifest(path, download)
 71    manifest = manifest[manifest["batch"].isin(batches) & manifest["file_role"].isin(["image", "segmentation_mask"])]
 72
 73    for file_path in manifest["Files"]:
 74        local_path = os.path.join(path, file_path)
 75        os.makedirs(os.path.dirname(local_path), exist_ok=True)
 76        util.download_source(local_path, f"{BASE_URL}/{file_path}", download, checksum=None)
 77
 78    return path
 79
 80
 81def get_imc_kidney_paths(
 82    path: Union[os.PathLike, str],
 83    batches: Optional[Sequence[str]] = None,
 84    download: bool = False,
 85) -> Tuple[List[str], List[str]]:
 86    """Get paths to the IMC kidney (AIN) images and cell instance segmentation masks.
 87
 88    Args:
 89        path: Filepath to a folder where the downloaded data will be saved.
 90        batches: The batch names to load. By default all 23 batches are loaded.
 91        download: Whether to download the data if it is not present.
 92
 93    Returns:
 94        The image paths and corresponding mask paths.
 95    """
 96    root = get_imc_kidney_data(path, batches, download)
 97    if batches is None:
 98        batches = BATCHES
 99
100    raw_paths, label_paths = [], []
101    for batch in batches:
102        raw_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "img", "*.tiff"))))
103        label_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "masks", "*.tiff"))))
104
105    missing_paths = [p for p in raw_paths + label_paths if not os.path.exists(p)]
106    if missing_paths:
107        raise RuntimeError(f"Could not find {len(missing_paths)} IMC kidney files.")
108
109    return raw_paths, label_paths
110
111
112def get_imc_kidney_dataset(
113    path: Union[os.PathLike, str],
114    patch_shape: Tuple[int, int],
115    batches: Optional[Sequence[str]] = None,
116    download: bool = False,
117    resize_inputs: bool = False,
118    **kwargs
119) -> Dataset:
120    """Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry.
121
122    Args:
123        path: Filepath to a folder where the downloaded data will be saved.
124        patch_shape: The 2D patch shape to use for training.
125        batches: The batch names to load. By default all 23 batches are loaded.
126        download: Whether to download the data if it is not present.
127        resize_inputs: Whether to resize the input images.
128        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
129
130    Returns:
131        The segmentation dataset.
132    """
133    if len(patch_shape) != 2:
134        raise ValueError(f"The IMC kidney patch shape must be two-dimensional, got {patch_shape}.")
135
136    raw_paths, label_paths = get_imc_kidney_paths(path, batches, download)
137
138    if resize_inputs:
139        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
140        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
141            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
142        )
143
144    return torch_em.default_segmentation_dataset(
145        raw_paths=raw_paths,
146        raw_key=None,
147        label_paths=label_paths,
148        label_key=None,
149        patch_shape=patch_shape,
150        is_seg_dataset=True,
151        with_channels=True,
152        ndim=2,
153        **kwargs
154    )
155
156
157def get_imc_kidney_loader(
158    path: Union[os.PathLike, str],
159    patch_shape: Tuple[int, int],
160    batch_size: int,
161    batches: Optional[Sequence[str]] = None,
162    download: bool = False,
163    resize_inputs: bool = False,
164    **kwargs
165) -> DataLoader:
166    """Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry.
167
168    Args:
169        path: Filepath to a folder where the downloaded data will be saved.
170        patch_shape: The 2D patch shape to use for training.
171        batch_size: The batch size for training.
172        batches: The batch names to load. By default all 23 batches are loaded.
173        download: Whether to download the data if it is not present.
174        resize_inputs: Whether to resize the input images.
175        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for
176            the PyTorch DataLoader.
177
178    Returns:
179        The DataLoader.
180    """
181    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
182    dataset = get_imc_kidney_dataset(
183        path, patch_shape, batches=batches, download=download, resize_inputs=resize_inputs, **ds_kwargs
184    )
185    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://ftp.ebi.ac.uk/pub/databases/biostudies/S-BIAD/694/S-BIAD3694/Files'
BATCHES = ('Batch1', 'Batch2', 'Batch3', 'Batch4', 'Batch5', 'Batch6', 'Batch7', 'Batch8', 'Batch9', 'Batch10', 'Batch11', 'Batch12', 'Batch13', 'Batch14', 'Batch15', 'Batch16', 'Batch17', 'Batch18', 'Batch19', 'Batch20', 'Batch21', 'Batch22', 'Batch23')
def get_imc_kidney_data( path: Union[os.PathLike, str], batches: Optional[Sequence[str]] = None, download: bool = False) -> str:
47def get_imc_kidney_data(
48    path: Union[os.PathLike, str],
49    batches: Optional[Sequence[str]] = None,
50    download: bool = False,
51) -> str:
52    """Download the IMC kidney (AIN) Discovery cohort data.
53
54    Args:
55        path: Filepath to a folder where the downloaded data will be saved.
56        batches: The batch names to prepare. By default all 23 batches are prepared, which
57            requires downloading several tens of gigabytes of data.
58        download: Whether to download the data if it is not present.
59
60    Returns:
61        Filepath to the folder where the data is stored.
62    """
63    if batches is None:
64        batches = BATCHES
65    else:
66        invalid = sorted(set(batches) - set(BATCHES))
67        if invalid:
68            raise ValueError(f"Invalid batch name(s) {invalid}. Choose from {BATCHES}.")
69
70    os.makedirs(path, exist_ok=True)
71    manifest = _get_manifest(path, download)
72    manifest = manifest[manifest["batch"].isin(batches) & manifest["file_role"].isin(["image", "segmentation_mask"])]
73
74    for file_path in manifest["Files"]:
75        local_path = os.path.join(path, file_path)
76        os.makedirs(os.path.dirname(local_path), exist_ok=True)
77        util.download_source(local_path, f"{BASE_URL}/{file_path}", download, checksum=None)
78
79    return path

Download the IMC kidney (AIN) Discovery cohort data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batches: The batch names to prepare. By default all 23 batches are prepared, which requires downloading several tens of gigabytes of data.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder where the data is stored.

def get_imc_kidney_paths( path: Union[os.PathLike, str], batches: Optional[Sequence[str]] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 82def get_imc_kidney_paths(
 83    path: Union[os.PathLike, str],
 84    batches: Optional[Sequence[str]] = None,
 85    download: bool = False,
 86) -> Tuple[List[str], List[str]]:
 87    """Get paths to the IMC kidney (AIN) images and cell instance segmentation masks.
 88
 89    Args:
 90        path: Filepath to a folder where the downloaded data will be saved.
 91        batches: The batch names to load. By default all 23 batches are loaded.
 92        download: Whether to download the data if it is not present.
 93
 94    Returns:
 95        The image paths and corresponding mask paths.
 96    """
 97    root = get_imc_kidney_data(path, batches, download)
 98    if batches is None:
 99        batches = BATCHES
100
101    raw_paths, label_paths = [], []
102    for batch in batches:
103        raw_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "img", "*.tiff"))))
104        label_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "masks", "*.tiff"))))
105
106    missing_paths = [p for p in raw_paths + label_paths if not os.path.exists(p)]
107    if missing_paths:
108        raise RuntimeError(f"Could not find {len(missing_paths)} IMC kidney files.")
109
110    return raw_paths, label_paths

Get paths to the IMC kidney (AIN) images and cell instance segmentation masks.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batches: The batch names to load. By default all 23 batches are loaded.
  • download: Whether to download the data if it is not present.
Returns:

The image paths and corresponding mask paths.

def get_imc_kidney_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batches: Optional[Sequence[str]] = None, download: bool = False, resize_inputs: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
113def get_imc_kidney_dataset(
114    path: Union[os.PathLike, str],
115    patch_shape: Tuple[int, int],
116    batches: Optional[Sequence[str]] = None,
117    download: bool = False,
118    resize_inputs: bool = False,
119    **kwargs
120) -> Dataset:
121    """Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry.
122
123    Args:
124        path: Filepath to a folder where the downloaded data will be saved.
125        patch_shape: The 2D patch shape to use for training.
126        batches: The batch names to load. By default all 23 batches are loaded.
127        download: Whether to download the data if it is not present.
128        resize_inputs: Whether to resize the input images.
129        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
130
131    Returns:
132        The segmentation dataset.
133    """
134    if len(patch_shape) != 2:
135        raise ValueError(f"The IMC kidney patch shape must be two-dimensional, got {patch_shape}.")
136
137    raw_paths, label_paths = get_imc_kidney_paths(path, batches, download)
138
139    if resize_inputs:
140        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
141        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
142            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
143        )
144
145    return torch_em.default_segmentation_dataset(
146        raw_paths=raw_paths,
147        raw_key=None,
148        label_paths=label_paths,
149        label_key=None,
150        patch_shape=patch_shape,
151        is_seg_dataset=True,
152        with_channels=True,
153        ndim=2,
154        **kwargs
155    )

Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The 2D patch shape to use for training.
  • batches: The batch names to load. By default all 23 batches are loaded.
  • download: Whether to download the data if it is not present.
  • resize_inputs: Whether to resize the input images.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_imc_kidney_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batch_size: int, batches: Optional[Sequence[str]] = None, download: bool = False, resize_inputs: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
158def get_imc_kidney_loader(
159    path: Union[os.PathLike, str],
160    patch_shape: Tuple[int, int],
161    batch_size: int,
162    batches: Optional[Sequence[str]] = None,
163    download: bool = False,
164    resize_inputs: bool = False,
165    **kwargs
166) -> DataLoader:
167    """Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry.
168
169    Args:
170        path: Filepath to a folder where the downloaded data will be saved.
171        patch_shape: The 2D patch shape to use for training.
172        batch_size: The batch size for training.
173        batches: The batch names to load. By default all 23 batches are loaded.
174        download: Whether to download the data if it is not present.
175        resize_inputs: Whether to resize the input images.
176        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for
177            the PyTorch DataLoader.
178
179    Returns:
180        The DataLoader.
181    """
182    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
183    dataset = get_imc_kidney_dataset(
184        path, patch_shape, batches=batches, download=download, resize_inputs=resize_inputs, **ds_kwargs
185    )
186    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The 2D patch shape to use for training.
  • batch_size: The batch size for training.
  • batches: The batch names to load. By default all 23 batches are loaded.
  • download: Whether to download the data if it is not present.
  • resize_inputs: Whether to resize the input images.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.