torch_em.data.datasets.light_microscopy.murine_phase_contrast
The Murine Phase-Contrast dataset contains annotations for cell instance segmentation in label-free phase-contrast microscopy images of three murine cell lines: OP9, MC3T3-E1 and NRG.
The dataset consists of 243 RGB images of 1920 x 1200 pixels, each showing a single cell line, with 16,567 manually annotated cell instances in total. The official split is done per image: 223 images (15,403 cells) for training and 20 images (1,164 cells) for validation. The annotations were made as LabelMe polygons, which the authors converted to instance masks with sequential polygon filling: a later polygon overwrites the overlap with an earlier one. This module uses these instance masks (one id per cell, 0 is background) and converts them once into tif files. A few polygons are completely overwritten by later ones, so the masks contain 15,371 (train) and 1,160 (val) visible instances instead of 15,403 and 1,164.
The archive additionally contains 16,567 single cell patches, analysis code, model checkpoints and evaluation tables for a cell type classification study. These are not used and are not extracted by this module.
NOTE: The Zenodo record states the CC-BY-4.0 license, while the README in the archive says that no license was assigned to its files. Check the terms before you redistribute the data.
The data is located at https://doi.org/10.5281/zenodo.21440736. Please cite the publication associated with the Zenodo record if you use this dataset in your research.
1"""The Murine Phase-Contrast dataset contains annotations for cell instance segmentation in label-free 2phase-contrast microscopy images of three murine cell lines: OP9, MC3T3-E1 and NRG. 3 4The dataset consists of 243 RGB images of 1920 x 1200 pixels, each showing a single cell line, with 516,567 manually annotated cell instances in total. The official split is done per image: 223 images 6(15,403 cells) for training and 20 images (1,164 cells) for validation. The annotations were made as LabelMe 7polygons, which the authors converted to instance masks with sequential polygon filling: a later polygon 8overwrites the overlap with an earlier one. This module uses these instance masks (one id per cell, 0 is 9background) and converts them once into tif files. A few polygons are completely overwritten by later ones, 10so the masks contain 15,371 (train) and 1,160 (val) visible instances instead of 15,403 and 1,164. 11 12The archive additionally contains 16,567 single cell patches, analysis code, model checkpoints and evaluation 13tables for a cell type classification study. These are not used and are not extracted by this module. 14 15NOTE: The Zenodo record states the CC-BY-4.0 license, while the README in the archive says that no license was 16assigned to its files. Check the terms before you redistribute the data. 17 18The data is located at https://doi.org/10.5281/zenodo.21440736. 19Please cite the publication associated with the Zenodo record if you use this dataset in your research. 20""" 21 22import os 23import csv 24import uuid 25import zipfile 26from natsort import natsorted 27from typing import Union, Tuple, Optional, List, Literal 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36URL = "https://zenodo.org/records/21440736/files/Xiong_cell_classification_dataset_v1.1_223_20_verified.zip" 37CHECKSUM = "303918988628135edb76de1b00b7e281c83b440e7cb34cda8c81a11d60f0fa5a" 38 39ROOT_IN_ZIP = "Xiong_cell_classification_dataset_v1.1_223_20_verified/dataset/" 40CELL_TYPES = ("OP9", "MC3T3-E1", "NRG") 41SPLITS = ("train", "val") 42 43 44def _convert_mask(npz_path, out_path): 45 import numpy as np 46 import tifffile 47 48 if os.path.exists(out_path): 49 return 50 51 label_map = np.load(npz_path)["label_map"].astype("uint16") 52 tmp_path = f"{out_path}.{uuid.uuid4().hex}.incomplete.tif" 53 tifffile.imwrite(tmp_path, label_map, compression="zlib") 54 os.replace(tmp_path, out_path) 55 56 57def _extract_needed_files(zip_path, data_dir): 58 prefixes = tuple(ROOT_IN_ZIP + folder + "/" for folder in ("raw_images", "instance_masks")) 59 wanted = ROOT_IN_ZIP + "dataset_splits/image_split.csv" 60 with zipfile.ZipFile(zip_path) as archive: 61 for member in archive.infolist(): 62 if member.is_dir() or not (member.filename.startswith(prefixes) or member.filename == wanted): 63 continue 64 target = os.path.join(data_dir, member.filename[len(ROOT_IN_ZIP):]) 65 os.makedirs(os.path.dirname(target), exist_ok=True) 66 tmp_path = f"{target}.{uuid.uuid4().hex}.incomplete" 67 with archive.open(member) as source, open(tmp_path, "wb") as sink: 68 sink.write(source.read()) 69 os.replace(tmp_path, target) 70 71 72def get_murine_phase_contrast_data(path: Union[os.PathLike, str], download: bool = False) -> str: 73 """Download the Murine Phase-Contrast dataset. 74 75 Args: 76 path: Filepath to a folder where the data is downloaded for further processing. 77 download: Whether to download the data if it is not present. 78 79 Returns: 80 Filepath where the data is downloaded. 81 """ 82 data_dir = os.path.join(path, "data") 83 if os.path.exists(os.path.join(data_dir, "dataset_splits", "image_split.csv")): 84 return data_dir 85 86 os.makedirs(path, exist_ok=True) 87 88 zip_path = os.path.join(path, "murine_phase_contrast.zip") 89 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 90 _extract_needed_files(zip_path, data_dir) 91 92 assert os.path.exists(os.path.join(data_dir, "dataset_splits", "image_split.csv")), \ 93 f"The extraction of the archive did not create the expected files in '{data_dir}'." 94 95 return data_dir 96 97 98def get_murine_phase_contrast_paths( 99 path: Union[os.PathLike, str], 100 split: Literal["train", "val"], 101 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 102 download: bool = False, 103) -> Tuple[List[str], List[str]]: 104 """Get paths to the Murine Phase-Contrast data. 105 106 Args: 107 path: Filepath to a folder where the data is downloaded for further processing. 108 split: The choice of data split. Either 'train' or 'val'. 109 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 110 download: Whether to download the data if it is not present. 111 112 Returns: 113 List of filepaths for the image data. 114 List of filepaths for the label data. 115 """ 116 if split not in SPLITS: 117 raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.") 118 if cell_type is not None and cell_type not in CELL_TYPES: 119 raise ValueError(f"'{cell_type}' is not a valid cell type. Choose one of {list(CELL_TYPES)}.") 120 121 data_dir = get_murine_phase_contrast_data(path, download) 122 label_dir = os.path.join(path, "labels") 123 os.makedirs(label_dir, exist_ok=True) 124 125 official_split = "validation" if split == "val" else "train" 126 with open(os.path.join(data_dir, "dataset_splits", "image_split.csv")) as f: 127 rows = [ 128 row for row in csv.DictReader(f) 129 if row["split"] == official_split and cell_type in (None, row["class_label"]) 130 ] 131 132 raw_paths, label_paths = [], [] 133 for row in natsorted(rows, key=lambda row: row["image_id"]): 134 raw_path = os.path.join(data_dir, "raw_images", row["filename"]) 135 npz_path = os.path.join(data_dir, "instance_masks", f"{row['image_id']}_instances.npz") 136 label_path = os.path.join(label_dir, f"{row['image_id']}.tif") 137 _convert_mask(npz_path, label_path) 138 139 raw_paths.append(raw_path) 140 label_paths.append(label_path) 141 142 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 143 assert all(os.path.exists(p) for p in raw_paths) 144 145 return raw_paths, label_paths 146 147 148def get_murine_phase_contrast_dataset( 149 path: Union[os.PathLike, str], 150 patch_shape: Tuple[int, int], 151 split: Literal["train", "val"], 152 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 153 resize_inputs: bool = False, 154 download: bool = False, 155 **kwargs 156) -> Dataset: 157 """Get the Murine Phase-Contrast dataset for cell instance segmentation in phase-contrast microscopy. 158 159 Args: 160 path: Filepath to a folder where the data is downloaded for further processing. 161 patch_shape: The patch shape to use for training. 162 split: The choice of data split. Either 'train' or 'val'. 163 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 164 resize_inputs: Whether to resize the inputs to the patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 167 168 Returns: 169 The segmentation dataset. 170 """ 171 raw_paths, label_paths = get_murine_phase_contrast_paths(path, split, cell_type, download) 172 173 if resize_inputs: 174 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 175 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 176 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 177 ) 178 179 return torch_em.default_segmentation_dataset( 180 raw_paths=raw_paths, 181 raw_key=None, 182 label_paths=label_paths, 183 label_key=None, 184 is_seg_dataset=False, 185 patch_shape=patch_shape, 186 **kwargs 187 ) 188 189 190def get_murine_phase_contrast_loader( 191 path: Union[os.PathLike, str], 192 batch_size: int, 193 patch_shape: Tuple[int, int], 194 split: Literal["train", "val"], 195 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 196 resize_inputs: bool = False, 197 download: bool = False, 198 **kwargs 199) -> DataLoader: 200 """Get the Murine Phase-Contrast dataloader for cell instance segmentation in phase-contrast microscopy. 201 202 Args: 203 path: Filepath to a folder where the data is downloaded for further processing. 204 batch_size: The batch size for training. 205 patch_shape: The patch shape to use for training. 206 split: The choice of data split. Either 'train' or 'val'. 207 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 208 resize_inputs: Whether to resize the inputs to the patch shape. 209 download: Whether to download the data if it is not present. 210 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 211 212 Returns: 213 The DataLoader. 214 """ 215 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 216 dataset = get_murine_phase_contrast_dataset( 217 path, patch_shape, split, cell_type, resize_inputs, download, **ds_kwargs 218 ) 219 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
73def get_murine_phase_contrast_data(path: Union[os.PathLike, str], download: bool = False) -> str: 74 """Download the Murine Phase-Contrast dataset. 75 76 Args: 77 path: Filepath to a folder where the data is downloaded for further processing. 78 download: Whether to download the data if it is not present. 79 80 Returns: 81 Filepath where the data is downloaded. 82 """ 83 data_dir = os.path.join(path, "data") 84 if os.path.exists(os.path.join(data_dir, "dataset_splits", "image_split.csv")): 85 return data_dir 86 87 os.makedirs(path, exist_ok=True) 88 89 zip_path = os.path.join(path, "murine_phase_contrast.zip") 90 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 91 _extract_needed_files(zip_path, data_dir) 92 93 assert os.path.exists(os.path.join(data_dir, "dataset_splits", "image_split.csv")), \ 94 f"The extraction of the archive did not create the expected files in '{data_dir}'." 95 96 return data_dir
Download the Murine Phase-Contrast dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
99def get_murine_phase_contrast_paths( 100 path: Union[os.PathLike, str], 101 split: Literal["train", "val"], 102 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 103 download: bool = False, 104) -> Tuple[List[str], List[str]]: 105 """Get paths to the Murine Phase-Contrast data. 106 107 Args: 108 path: Filepath to a folder where the data is downloaded for further processing. 109 split: The choice of data split. Either 'train' or 'val'. 110 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 111 download: Whether to download the data if it is not present. 112 113 Returns: 114 List of filepaths for the image data. 115 List of filepaths for the label data. 116 """ 117 if split not in SPLITS: 118 raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.") 119 if cell_type is not None and cell_type not in CELL_TYPES: 120 raise ValueError(f"'{cell_type}' is not a valid cell type. Choose one of {list(CELL_TYPES)}.") 121 122 data_dir = get_murine_phase_contrast_data(path, download) 123 label_dir = os.path.join(path, "labels") 124 os.makedirs(label_dir, exist_ok=True) 125 126 official_split = "validation" if split == "val" else "train" 127 with open(os.path.join(data_dir, "dataset_splits", "image_split.csv")) as f: 128 rows = [ 129 row for row in csv.DictReader(f) 130 if row["split"] == official_split and cell_type in (None, row["class_label"]) 131 ] 132 133 raw_paths, label_paths = [], [] 134 for row in natsorted(rows, key=lambda row: row["image_id"]): 135 raw_path = os.path.join(data_dir, "raw_images", row["filename"]) 136 npz_path = os.path.join(data_dir, "instance_masks", f"{row['image_id']}_instances.npz") 137 label_path = os.path.join(label_dir, f"{row['image_id']}.tif") 138 _convert_mask(npz_path, label_path) 139 140 raw_paths.append(raw_path) 141 label_paths.append(label_path) 142 143 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 144 assert all(os.path.exists(p) for p in raw_paths) 145 146 return raw_paths, label_paths
Get paths to the Murine Phase-Contrast data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. Either 'train' or 'val'.
- cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
149def get_murine_phase_contrast_dataset( 150 path: Union[os.PathLike, str], 151 patch_shape: Tuple[int, int], 152 split: Literal["train", "val"], 153 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> Dataset: 158 """Get the Murine Phase-Contrast dataset for cell instance segmentation in phase-contrast microscopy. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 patch_shape: The patch shape to use for training. 163 split: The choice of data split. Either 'train' or 'val'. 164 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 165 resize_inputs: Whether to resize the inputs to the patch shape. 166 download: Whether to download the data if it is not present. 167 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 168 169 Returns: 170 The segmentation dataset. 171 """ 172 raw_paths, label_paths = get_murine_phase_contrast_paths(path, split, cell_type, download) 173 174 if resize_inputs: 175 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 176 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 177 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 178 ) 179 180 return torch_em.default_segmentation_dataset( 181 raw_paths=raw_paths, 182 raw_key=None, 183 label_paths=label_paths, 184 label_key=None, 185 is_seg_dataset=False, 186 patch_shape=patch_shape, 187 **kwargs 188 )
Get the Murine Phase-Contrast dataset for cell instance segmentation in phase-contrast microscopy.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'val'.
- cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
191def get_murine_phase_contrast_loader( 192 path: Union[os.PathLike, str], 193 batch_size: int, 194 patch_shape: Tuple[int, int], 195 split: Literal["train", "val"], 196 cell_type: Optional[Literal["OP9", "MC3T3-E1", "NRG"]] = None, 197 resize_inputs: bool = False, 198 download: bool = False, 199 **kwargs 200) -> DataLoader: 201 """Get the Murine Phase-Contrast dataloader for cell instance segmentation in phase-contrast microscopy. 202 203 Args: 204 path: Filepath to a folder where the data is downloaded for further processing. 205 batch_size: The batch size for training. 206 patch_shape: The patch shape to use for training. 207 split: The choice of data split. Either 'train' or 'val'. 208 cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used. 209 resize_inputs: Whether to resize the inputs to the patch shape. 210 download: Whether to download the data if it is not present. 211 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 212 213 Returns: 214 The DataLoader. 215 """ 216 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 217 dataset = get_murine_phase_contrast_dataset( 218 path, patch_shape, split, cell_type, resize_inputs, download, **ds_kwargs 219 ) 220 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Murine Phase-Contrast dataloader for cell instance segmentation in phase-contrast microscopy.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. Either 'train' or 'val'.
- cell_type: The choice of cell line. One of 'OP9', 'MC3T3-E1' or 'NRG'. By default all are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.