torch_em.data.datasets.histopathology.imc_kidney
This dataset contains cell instance segmentation annotations for imaging mass cytometry (IMC) of human kidney biopsies with acute interstitial nephritis (AIN), acute tubular injury (ATI), and histologically normal reference tissue.
The data is from the publication "Spatial analysis reveals cellular microenvironments and mechanisms of inflammation and injury in acute interstitial nephritis" and hosted on the BioImage Archive at https://www.ebi.ac.uk/biostudies/bioimages/studies/S-BIAD3694. It is available under the CC0 license. Please cite it if you use this dataset in your research.
This loader covers the Discovery cohort (23 experimental batches, 325 acquisitions), which is provided as processed multichannel TIFF stacks (35-marker antibody panel) with matching per-cell instance segmentation masks generated with Mesmer. The BioImage Archive record also hosts a Validation cohort of raw Hyperion text exports without segmentation masks, which is out of scope for this loader.
1"""This dataset contains cell instance segmentation annotations for imaging mass cytometry (IMC) 2of human kidney biopsies with acute interstitial nephritis (AIN), acute tubular injury (ATI), and 3histologically normal reference tissue. 4 5The data is from the publication "Spatial analysis reveals cellular microenvironments and 6mechanisms of inflammation and injury in acute interstitial nephritis" and hosted on the 7BioImage Archive at https://www.ebi.ac.uk/biostudies/bioimages/studies/S-BIAD3694. It is 8available under the CC0 license. Please cite it if you use this dataset in your research. 9 10This loader covers the Discovery cohort (23 experimental batches, 325 acquisitions), which is 11provided as processed multichannel TIFF stacks (35-marker antibody panel) with matching per-cell 12instance segmentation masks generated with Mesmer. The BioImage Archive record also hosts a 13Validation cohort of raw Hyperion text exports without segmentation masks, which is out of scope 14for this loader. 15""" 16 17import os 18from glob import glob 19from typing import List, Optional, Sequence, Tuple, Union 20 21import pandas as pd 22 23from torch.utils.data import Dataset, DataLoader 24 25import torch_em 26 27from .. import util 28 29 30BASE_URL = "https://ftp.ebi.ac.uk/pub/databases/biostudies/S-BIAD/694/S-BIAD3694/Files" 31 32BATCHES = tuple(f"Batch{i}" for i in range(1, 24)) 33 34 35def _get_manifest(path, download): 36 # NOTE: The per-batch 'images.csv' manifests list acquisitions that were planned but not all of 37 # them were actually deposited (e.g. excluded ROIs). The top-level file list only contains the 38 # acquisitions that were actually uploaded, so we rely on it to determine the real image / mask pairs. 39 manifest_path = os.path.join(path, "bia_filelist_all.tsv") 40 os.makedirs(path, exist_ok=True) 41 util.download_source(manifest_path, f"{BASE_URL}/bia_filelist_all.tsv", download, checksum=None) 42 manifest = pd.read_csv(manifest_path, sep="\t") 43 return manifest[manifest["cohort"] == "discovery"] 44 45 46def get_imc_kidney_data( 47 path: Union[os.PathLike, str], 48 batches: Optional[Sequence[str]] = None, 49 download: bool = False, 50) -> str: 51 """Download the IMC kidney (AIN) Discovery cohort data. 52 53 Args: 54 path: Filepath to a folder where the downloaded data will be saved. 55 batches: The batch names to prepare. By default all 23 batches are prepared, which 56 requires downloading several tens of gigabytes of data. 57 download: Whether to download the data if it is not present. 58 59 Returns: 60 Filepath to the folder where the data is stored. 61 """ 62 if batches is None: 63 batches = BATCHES 64 else: 65 invalid = sorted(set(batches) - set(BATCHES)) 66 if invalid: 67 raise ValueError(f"Invalid batch name(s) {invalid}. Choose from {BATCHES}.") 68 69 os.makedirs(path, exist_ok=True) 70 manifest = _get_manifest(path, download) 71 manifest = manifest[manifest["batch"].isin(batches) & manifest["file_role"].isin(["image", "segmentation_mask"])] 72 73 for file_path in manifest["Files"]: 74 local_path = os.path.join(path, file_path) 75 os.makedirs(os.path.dirname(local_path), exist_ok=True) 76 util.download_source(local_path, f"{BASE_URL}/{file_path}", download, checksum=None) 77 78 return path 79 80 81def get_imc_kidney_paths( 82 path: Union[os.PathLike, str], 83 batches: Optional[Sequence[str]] = None, 84 download: bool = False, 85) -> Tuple[List[str], List[str]]: 86 """Get paths to the IMC kidney (AIN) images and cell instance segmentation masks. 87 88 Args: 89 path: Filepath to a folder where the downloaded data will be saved. 90 batches: The batch names to load. By default all 23 batches are loaded. 91 download: Whether to download the data if it is not present. 92 93 Returns: 94 The image paths and corresponding mask paths. 95 """ 96 root = get_imc_kidney_data(path, batches, download) 97 if batches is None: 98 batches = BATCHES 99 100 raw_paths, label_paths = [], [] 101 for batch in batches: 102 raw_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "img", "*.tiff")))) 103 label_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "masks", "*.tiff")))) 104 105 missing_paths = [p for p in raw_paths + label_paths if not os.path.exists(p)] 106 if missing_paths: 107 raise RuntimeError(f"Could not find {len(missing_paths)} IMC kidney files.") 108 109 return raw_paths, label_paths 110 111 112def get_imc_kidney_dataset( 113 path: Union[os.PathLike, str], 114 patch_shape: Tuple[int, int], 115 batches: Optional[Sequence[str]] = None, 116 download: bool = False, 117 resize_inputs: bool = False, 118 **kwargs 119) -> Dataset: 120 """Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry. 121 122 Args: 123 path: Filepath to a folder where the downloaded data will be saved. 124 patch_shape: The 2D patch shape to use for training. 125 batches: The batch names to load. By default all 23 batches are loaded. 126 download: Whether to download the data if it is not present. 127 resize_inputs: Whether to resize the input images. 128 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 129 130 Returns: 131 The segmentation dataset. 132 """ 133 if len(patch_shape) != 2: 134 raise ValueError(f"The IMC kidney patch shape must be two-dimensional, got {patch_shape}.") 135 136 raw_paths, label_paths = get_imc_kidney_paths(path, batches, download) 137 138 if resize_inputs: 139 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 140 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 141 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 142 ) 143 144 return torch_em.default_segmentation_dataset( 145 raw_paths=raw_paths, 146 raw_key=None, 147 label_paths=label_paths, 148 label_key=None, 149 patch_shape=patch_shape, 150 is_seg_dataset=True, 151 with_channels=True, 152 ndim=2, 153 **kwargs 154 ) 155 156 157def get_imc_kidney_loader( 158 path: Union[os.PathLike, str], 159 patch_shape: Tuple[int, int], 160 batch_size: int, 161 batches: Optional[Sequence[str]] = None, 162 download: bool = False, 163 resize_inputs: bool = False, 164 **kwargs 165) -> DataLoader: 166 """Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry. 167 168 Args: 169 path: Filepath to a folder where the downloaded data will be saved. 170 patch_shape: The 2D patch shape to use for training. 171 batch_size: The batch size for training. 172 batches: The batch names to load. By default all 23 batches are loaded. 173 download: Whether to download the data if it is not present. 174 resize_inputs: Whether to resize the input images. 175 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for 176 the PyTorch DataLoader. 177 178 Returns: 179 The DataLoader. 180 """ 181 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 182 dataset = get_imc_kidney_dataset( 183 path, patch_shape, batches=batches, download=download, resize_inputs=resize_inputs, **ds_kwargs 184 ) 185 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
47def get_imc_kidney_data( 48 path: Union[os.PathLike, str], 49 batches: Optional[Sequence[str]] = None, 50 download: bool = False, 51) -> str: 52 """Download the IMC kidney (AIN) Discovery cohort data. 53 54 Args: 55 path: Filepath to a folder where the downloaded data will be saved. 56 batches: The batch names to prepare. By default all 23 batches are prepared, which 57 requires downloading several tens of gigabytes of data. 58 download: Whether to download the data if it is not present. 59 60 Returns: 61 Filepath to the folder where the data is stored. 62 """ 63 if batches is None: 64 batches = BATCHES 65 else: 66 invalid = sorted(set(batches) - set(BATCHES)) 67 if invalid: 68 raise ValueError(f"Invalid batch name(s) {invalid}. Choose from {BATCHES}.") 69 70 os.makedirs(path, exist_ok=True) 71 manifest = _get_manifest(path, download) 72 manifest = manifest[manifest["batch"].isin(batches) & manifest["file_role"].isin(["image", "segmentation_mask"])] 73 74 for file_path in manifest["Files"]: 75 local_path = os.path.join(path, file_path) 76 os.makedirs(os.path.dirname(local_path), exist_ok=True) 77 util.download_source(local_path, f"{BASE_URL}/{file_path}", download, checksum=None) 78 79 return path
Download the IMC kidney (AIN) Discovery cohort data.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- batches: The batch names to prepare. By default all 23 batches are prepared, which requires downloading several tens of gigabytes of data.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder where the data is stored.
82def get_imc_kidney_paths( 83 path: Union[os.PathLike, str], 84 batches: Optional[Sequence[str]] = None, 85 download: bool = False, 86) -> Tuple[List[str], List[str]]: 87 """Get paths to the IMC kidney (AIN) images and cell instance segmentation masks. 88 89 Args: 90 path: Filepath to a folder where the downloaded data will be saved. 91 batches: The batch names to load. By default all 23 batches are loaded. 92 download: Whether to download the data if it is not present. 93 94 Returns: 95 The image paths and corresponding mask paths. 96 """ 97 root = get_imc_kidney_data(path, batches, download) 98 if batches is None: 99 batches = BATCHES 100 101 raw_paths, label_paths = [], [] 102 for batch in batches: 103 raw_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "img", "*.tiff")))) 104 label_paths.extend(sorted(glob(os.path.join(root, "Discovery", batch, "masks", "*.tiff")))) 105 106 missing_paths = [p for p in raw_paths + label_paths if not os.path.exists(p)] 107 if missing_paths: 108 raise RuntimeError(f"Could not find {len(missing_paths)} IMC kidney files.") 109 110 return raw_paths, label_paths
Get paths to the IMC kidney (AIN) images and cell instance segmentation masks.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- batches: The batch names to load. By default all 23 batches are loaded.
- download: Whether to download the data if it is not present.
Returns:
The image paths and corresponding mask paths.
113def get_imc_kidney_dataset( 114 path: Union[os.PathLike, str], 115 patch_shape: Tuple[int, int], 116 batches: Optional[Sequence[str]] = None, 117 download: bool = False, 118 resize_inputs: bool = False, 119 **kwargs 120) -> Dataset: 121 """Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry. 122 123 Args: 124 path: Filepath to a folder where the downloaded data will be saved. 125 patch_shape: The 2D patch shape to use for training. 126 batches: The batch names to load. By default all 23 batches are loaded. 127 download: Whether to download the data if it is not present. 128 resize_inputs: Whether to resize the input images. 129 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 130 131 Returns: 132 The segmentation dataset. 133 """ 134 if len(patch_shape) != 2: 135 raise ValueError(f"The IMC kidney patch shape must be two-dimensional, got {patch_shape}.") 136 137 raw_paths, label_paths = get_imc_kidney_paths(path, batches, download) 138 139 if resize_inputs: 140 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 141 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 142 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 143 ) 144 145 return torch_em.default_segmentation_dataset( 146 raw_paths=raw_paths, 147 raw_key=None, 148 label_paths=label_paths, 149 label_key=None, 150 patch_shape=patch_shape, 151 is_seg_dataset=True, 152 with_channels=True, 153 ndim=2, 154 **kwargs 155 )
Get the IMC kidney (AIN) dataset for cell instance segmentation in imaging mass cytometry.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- patch_shape: The 2D patch shape to use for training.
- batches: The batch names to load. By default all 23 batches are loaded.
- download: Whether to download the data if it is not present.
- resize_inputs: Whether to resize the input images.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
158def get_imc_kidney_loader( 159 path: Union[os.PathLike, str], 160 patch_shape: Tuple[int, int], 161 batch_size: int, 162 batches: Optional[Sequence[str]] = None, 163 download: bool = False, 164 resize_inputs: bool = False, 165 **kwargs 166) -> DataLoader: 167 """Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry. 168 169 Args: 170 path: Filepath to a folder where the downloaded data will be saved. 171 patch_shape: The 2D patch shape to use for training. 172 batch_size: The batch size for training. 173 batches: The batch names to load. By default all 23 batches are loaded. 174 download: Whether to download the data if it is not present. 175 resize_inputs: Whether to resize the input images. 176 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for 177 the PyTorch DataLoader. 178 179 Returns: 180 The DataLoader. 181 """ 182 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 183 dataset = get_imc_kidney_dataset( 184 path, patch_shape, batches=batches, download=download, resize_inputs=resize_inputs, **ds_kwargs 185 ) 186 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the IMC kidney (AIN) dataloader for cell instance segmentation in imaging mass cytometry.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- patch_shape: The 2D patch shape to use for training.
- batch_size: The batch size for training.
- batches: The batch names to load. By default all 23 batches are loaded.
- download: Whether to download the data if it is not present.
- resize_inputs: Whether to resize the input images.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.