torch_em.data.datasets.light_microscopy.pyropia
The Pyropia dataset contains annotations for cell instance segmentation in high-resolution microscopy images of the red alga Pyropia haitanensis (also called Porphyra haitanensis).
The dataset consists of 32 RGB images (2160 x 3840 pixels) of two strains, 'Sansha-5' and 'WO115-2'.
Each strain has 8 experimental subsets (plant_1 to plant_8 and plant_9 to plant_16), which each
contain the same tissue imaged at 0 h and 24 h. Cells were annotated manually with LabelMe polygons,
which gives 3,359 cell instances in total. The annotation files also encode 1,255 tracking relationships
between the two time points and 283 cell divisions, which are not used by this loader.
This module rasterizes the polygons into instance label images (one id per cell, in the order of the
polygons in the annotation file). Overlaps between polygons are negligible, later polygons overwrite earlier ones.
The data is available on two Zenodo records with identical images and annotations: https://zenodo.org/records/19571835 (used by this loader, includes a README and one folder per subset) and https://zenodo.org/records/20301518 (the same files sorted into one folder per strain). It is released under a CC-BY-4.0 license.
Please cite the publication associated with the Zenodo record if you use this dataset in your research.
1"""The Pyropia dataset contains annotations for cell instance segmentation in high-resolution 2microscopy images of the red alga Pyropia haitanensis (also called Porphyra haitanensis). 3 4The dataset consists of 32 RGB images (2160 x 3840 pixels) of two strains, 'Sansha-5' and 'WO115-2'. 5Each strain has 8 experimental subsets (`plant_1` to `plant_8` and `plant_9` to `plant_16`), which each 6contain the same tissue imaged at 0 h and 24 h. Cells were annotated manually with LabelMe polygons, 7which gives 3,359 cell instances in total. The annotation files also encode 1,255 tracking relationships 8between the two time points and 283 cell divisions, which are not used by this loader. 9This module rasterizes the polygons into instance label images (one id per cell, in the order of the 10polygons in the annotation file). Overlaps between polygons are negligible, later polygons overwrite earlier ones. 11 12The data is available on two Zenodo records with identical images and annotations: 13https://zenodo.org/records/19571835 (used by this loader, includes a README and one folder per subset) 14and https://zenodo.org/records/20301518 (the same files sorted into one folder per strain). 15It is released under a CC-BY-4.0 license. 16 17Please cite the publication associated with the Zenodo record if you use this dataset in your research. 18""" 19 20import os 21import json 22import uuid 23from glob import glob 24from natsort import natsorted 25from typing import Union, Tuple, Optional, List, Literal 26 27from torch.utils.data import Dataset, DataLoader 28 29import torch_em 30 31from .. import util 32 33 34URL = "https://zenodo.org/records/19571835/files/data.zip" 35CHECKSUM = "55aa506e6fe6a68582f85ef5ac084a4107c686ae231a89ee9c3ae08f35558ecb" 36 37STRAINS = {"sansha-5": range(1, 9), "wo115-2": range(9, 17)} 38 39 40def _rasterize_annotation(json_path, out_path): 41 import numpy as np 42 import tifffile 43 from skimage.draw import polygon 44 45 if os.path.exists(out_path): 46 return 47 48 with open(json_path) as f: 49 annotation = json.load(f) 50 51 height, width = annotation["imageHeight"], annotation["imageWidth"] 52 labels = np.zeros((height, width), dtype="uint16") 53 for instance_id, shape in enumerate(annotation["shapes"], start=1): 54 points = np.array(shape["points"]) 55 rr, cc = polygon(points[:, 1], points[:, 0], (height, width)) 56 labels[rr, cc] = instance_id 57 58 tmp_path = f"{out_path}.{uuid.uuid4().hex}.incomplete.tif" 59 tifffile.imwrite(tmp_path, labels, compression="zlib") 60 os.replace(tmp_path, out_path) 61 62 63def get_pyropia_data(path: Union[os.PathLike, str], download: bool = False) -> str: 64 """Download the Pyropia dataset. 65 66 Args: 67 path: Filepath to a folder where the data is downloaded for further processing. 68 download: Whether to download the data if it is not present. 69 70 Returns: 71 Filepath where the data is downloaded. 72 """ 73 data_dir = os.path.join(path, "data") 74 if os.path.exists(data_dir): 75 return data_dir 76 77 os.makedirs(path, exist_ok=True) 78 79 zip_path = os.path.join(path, "pyropia.zip") 80 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 81 util.unzip(zip_path=zip_path, dst=path, remove=False) 82 83 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'." 84 85 return data_dir 86 87 88def get_pyropia_paths( 89 path: Union[os.PathLike, str], strain: Optional[Literal["sansha-5", "wo115-2"]] = None, download: bool = False, 90) -> Tuple[List[str], List[str]]: 91 """Get paths to the Pyropia data. 92 93 Args: 94 path: Filepath to a folder where the data is downloaded for further processing. 95 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 96 download: Whether to download the data if it is not present. 97 98 Returns: 99 List of filepaths for the image data. 100 List of filepaths for the label data. 101 """ 102 if strain is not None and strain not in STRAINS: 103 raise ValueError(f"'{strain}' is not a valid strain. Choose one of {list(STRAINS)}.") 104 105 data_dir = get_pyropia_data(path, download) 106 label_dir = os.path.join(path, "labels") 107 os.makedirs(label_dir, exist_ok=True) 108 109 plants = [i for name, ids in STRAINS.items() if strain in (None, name) for i in ids] 110 111 raw_paths, label_paths = [], [] 112 for plant in plants: 113 json_paths = natsorted(glob(os.path.join(data_dir, f"plant_{plant}", "json", "*.json"))) 114 for json_path in json_paths: 115 stem = os.path.splitext(os.path.basename(json_path))[0] 116 raw_path = glob(os.path.join(data_dir, f"plant_{plant}", "img", f"{stem}.tif*")) 117 assert len(raw_path) == 1, f"Expected one image for '{json_path}', found {len(raw_path)}." 118 119 label_path = os.path.join(label_dir, f"plant_{plant}_{stem}.tif") 120 _rasterize_annotation(json_path, label_path) 121 122 raw_paths.append(raw_path[0]) 123 label_paths.append(label_path) 124 125 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 126 127 return raw_paths, label_paths 128 129 130def get_pyropia_dataset( 131 path: Union[os.PathLike, str], 132 patch_shape: Tuple[int, int], 133 strain: Optional[Literal["sansha-5", "wo115-2"]] = None, 134 resize_inputs: bool = False, 135 download: bool = False, 136 **kwargs 137) -> Dataset: 138 """Get the Pyropia dataset for cell instance segmentation in plant microscopy images. 139 140 Args: 141 path: Filepath to a folder where the data is downloaded for further processing. 142 patch_shape: The patch shape to use for training. 143 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 144 resize_inputs: Whether to resize the inputs to the patch shape. 145 download: Whether to download the data if it is not present. 146 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 147 148 Returns: 149 The segmentation dataset. 150 """ 151 raw_paths, label_paths = get_pyropia_paths(path, strain, download) 152 153 if resize_inputs: 154 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 155 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 156 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 157 ) 158 159 return torch_em.default_segmentation_dataset( 160 raw_paths=raw_paths, 161 raw_key=None, 162 label_paths=label_paths, 163 label_key=None, 164 is_seg_dataset=False, 165 patch_shape=patch_shape, 166 **kwargs 167 ) 168 169 170def get_pyropia_loader( 171 path: Union[os.PathLike, str], 172 batch_size: int, 173 patch_shape: Tuple[int, int], 174 strain: Optional[Literal["sansha-5", "wo115-2"]] = None, 175 resize_inputs: bool = False, 176 download: bool = False, 177 **kwargs 178) -> DataLoader: 179 """Get the Pyropia dataloader for cell instance segmentation in plant microscopy images. 180 181 Args: 182 path: Filepath to a folder where the data is downloaded for further processing. 183 batch_size: The batch size for training. 184 patch_shape: The patch shape to use for training. 185 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 186 resize_inputs: Whether to resize the inputs to the patch shape. 187 download: Whether to download the data if it is not present. 188 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 189 190 Returns: 191 The DataLoader. 192 """ 193 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 194 dataset = get_pyropia_dataset(path, patch_shape, strain, resize_inputs, download, **ds_kwargs) 195 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
64def get_pyropia_data(path: Union[os.PathLike, str], download: bool = False) -> str: 65 """Download the Pyropia dataset. 66 67 Args: 68 path: Filepath to a folder where the data is downloaded for further processing. 69 download: Whether to download the data if it is not present. 70 71 Returns: 72 Filepath where the data is downloaded. 73 """ 74 data_dir = os.path.join(path, "data") 75 if os.path.exists(data_dir): 76 return data_dir 77 78 os.makedirs(path, exist_ok=True) 79 80 zip_path = os.path.join(path, "pyropia.zip") 81 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 82 util.unzip(zip_path=zip_path, dst=path, remove=False) 83 84 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'." 85 86 return data_dir
Download the Pyropia dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
89def get_pyropia_paths( 90 path: Union[os.PathLike, str], strain: Optional[Literal["sansha-5", "wo115-2"]] = None, download: bool = False, 91) -> Tuple[List[str], List[str]]: 92 """Get paths to the Pyropia data. 93 94 Args: 95 path: Filepath to a folder where the data is downloaded for further processing. 96 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 97 download: Whether to download the data if it is not present. 98 99 Returns: 100 List of filepaths for the image data. 101 List of filepaths for the label data. 102 """ 103 if strain is not None and strain not in STRAINS: 104 raise ValueError(f"'{strain}' is not a valid strain. Choose one of {list(STRAINS)}.") 105 106 data_dir = get_pyropia_data(path, download) 107 label_dir = os.path.join(path, "labels") 108 os.makedirs(label_dir, exist_ok=True) 109 110 plants = [i for name, ids in STRAINS.items() if strain in (None, name) for i in ids] 111 112 raw_paths, label_paths = [], [] 113 for plant in plants: 114 json_paths = natsorted(glob(os.path.join(data_dir, f"plant_{plant}", "json", "*.json"))) 115 for json_path in json_paths: 116 stem = os.path.splitext(os.path.basename(json_path))[0] 117 raw_path = glob(os.path.join(data_dir, f"plant_{plant}", "img", f"{stem}.tif*")) 118 assert len(raw_path) == 1, f"Expected one image for '{json_path}', found {len(raw_path)}." 119 120 label_path = os.path.join(label_dir, f"plant_{plant}_{stem}.tif") 121 _rasterize_annotation(json_path, label_path) 122 123 raw_paths.append(raw_path[0]) 124 label_paths.append(label_path) 125 126 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 127 128 return raw_paths, label_paths
Get paths to the Pyropia data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
131def get_pyropia_dataset( 132 path: Union[os.PathLike, str], 133 patch_shape: Tuple[int, int], 134 strain: Optional[Literal["sansha-5", "wo115-2"]] = None, 135 resize_inputs: bool = False, 136 download: bool = False, 137 **kwargs 138) -> Dataset: 139 """Get the Pyropia dataset for cell instance segmentation in plant microscopy images. 140 141 Args: 142 path: Filepath to a folder where the data is downloaded for further processing. 143 patch_shape: The patch shape to use for training. 144 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 145 resize_inputs: Whether to resize the inputs to the patch shape. 146 download: Whether to download the data if it is not present. 147 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 148 149 Returns: 150 The segmentation dataset. 151 """ 152 raw_paths, label_paths = get_pyropia_paths(path, strain, download) 153 154 if resize_inputs: 155 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 156 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 157 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 158 ) 159 160 return torch_em.default_segmentation_dataset( 161 raw_paths=raw_paths, 162 raw_key=None, 163 label_paths=label_paths, 164 label_key=None, 165 is_seg_dataset=False, 166 patch_shape=patch_shape, 167 **kwargs 168 )
Get the Pyropia dataset for cell instance segmentation in plant microscopy images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
171def get_pyropia_loader( 172 path: Union[os.PathLike, str], 173 batch_size: int, 174 patch_shape: Tuple[int, int], 175 strain: Optional[Literal["sansha-5", "wo115-2"]] = None, 176 resize_inputs: bool = False, 177 download: bool = False, 178 **kwargs 179) -> DataLoader: 180 """Get the Pyropia dataloader for cell instance segmentation in plant microscopy images. 181 182 Args: 183 path: Filepath to a folder where the data is downloaded for further processing. 184 batch_size: The batch size for training. 185 patch_shape: The patch shape to use for training. 186 strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used. 187 resize_inputs: Whether to resize the inputs to the patch shape. 188 download: Whether to download the data if it is not present. 189 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 190 191 Returns: 192 The DataLoader. 193 """ 194 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 195 dataset = get_pyropia_dataset(path, patch_shape, strain, resize_inputs, download, **ds_kwargs) 196 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Pyropia dataloader for cell instance segmentation in plant microscopy images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.