torch_em.data.datasets.light_microscopy.pyropia

The Pyropia dataset contains annotations for cell instance segmentation in high-resolution microscopy images of the red alga Pyropia haitanensis (also called Porphyra haitanensis).

The dataset consists of 32 RGB images (2160 x 3840 pixels) of two strains, 'Sansha-5' and 'WO115-2'. Each strain has 8 experimental subsets (plant_1 to plant_8 and plant_9 to plant_16), which each contain the same tissue imaged at 0 h and 24 h. Cells were annotated manually with LabelMe polygons, which gives 3,359 cell instances in total. The annotation files also encode 1,255 tracking relationships between the two time points and 283 cell divisions, which are not used by this loader. This module rasterizes the polygons into instance label images (one id per cell, in the order of the polygons in the annotation file). Overlaps between polygons are negligible, later polygons overwrite earlier ones.

The data is available on two Zenodo records with identical images and annotations: https://zenodo.org/records/19571835 (used by this loader, includes a README and one folder per subset) and https://zenodo.org/records/20301518 (the same files sorted into one folder per strain). It is released under a CC-BY-4.0 license.

Please cite the publication associated with the Zenodo record if you use this dataset in your research.

  1"""The Pyropia dataset contains annotations for cell instance segmentation in high-resolution
  2microscopy images of the red alga Pyropia haitanensis (also called Porphyra haitanensis).
  3
  4The dataset consists of 32 RGB images (2160 x 3840 pixels) of two strains, 'Sansha-5' and 'WO115-2'.
  5Each strain has 8 experimental subsets (`plant_1` to `plant_8` and `plant_9` to `plant_16`), which each
  6contain the same tissue imaged at 0 h and 24 h. Cells were annotated manually with LabelMe polygons,
  7which gives 3,359 cell instances in total. The annotation files also encode 1,255 tracking relationships
  8between the two time points and 283 cell divisions, which are not used by this loader.
  9This module rasterizes the polygons into instance label images (one id per cell, in the order of the
 10polygons in the annotation file). Overlaps between polygons are negligible, later polygons overwrite earlier ones.
 11
 12The data is available on two Zenodo records with identical images and annotations:
 13https://zenodo.org/records/19571835 (used by this loader, includes a README and one folder per subset)
 14and https://zenodo.org/records/20301518 (the same files sorted into one folder per strain).
 15It is released under a CC-BY-4.0 license.
 16
 17Please cite the publication associated with the Zenodo record if you use this dataset in your research.
 18"""
 19
 20import os
 21import json
 22import uuid
 23from glob import glob
 24from natsort import natsorted
 25from typing import Union, Tuple, Optional, List, Literal
 26
 27from torch.utils.data import Dataset, DataLoader
 28
 29import torch_em
 30
 31from .. import util
 32
 33
 34URL = "https://zenodo.org/records/19571835/files/data.zip"
 35CHECKSUM = "55aa506e6fe6a68582f85ef5ac084a4107c686ae231a89ee9c3ae08f35558ecb"
 36
 37STRAINS = {"sansha-5": range(1, 9), "wo115-2": range(9, 17)}
 38
 39
 40def _rasterize_annotation(json_path, out_path):
 41    import numpy as np
 42    import tifffile
 43    from skimage.draw import polygon
 44
 45    if os.path.exists(out_path):
 46        return
 47
 48    with open(json_path) as f:
 49        annotation = json.load(f)
 50
 51    height, width = annotation["imageHeight"], annotation["imageWidth"]
 52    labels = np.zeros((height, width), dtype="uint16")
 53    for instance_id, shape in enumerate(annotation["shapes"], start=1):
 54        points = np.array(shape["points"])
 55        rr, cc = polygon(points[:, 1], points[:, 0], (height, width))
 56        labels[rr, cc] = instance_id
 57
 58    tmp_path = f"{out_path}.{uuid.uuid4().hex}.incomplete.tif"
 59    tifffile.imwrite(tmp_path, labels, compression="zlib")
 60    os.replace(tmp_path, out_path)
 61
 62
 63def get_pyropia_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 64    """Download the Pyropia dataset.
 65
 66    Args:
 67        path: Filepath to a folder where the data is downloaded for further processing.
 68        download: Whether to download the data if it is not present.
 69
 70    Returns:
 71        Filepath where the data is downloaded.
 72    """
 73    data_dir = os.path.join(path, "data")
 74    if os.path.exists(data_dir):
 75        return data_dir
 76
 77    os.makedirs(path, exist_ok=True)
 78
 79    zip_path = os.path.join(path, "pyropia.zip")
 80    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 81    util.unzip(zip_path=zip_path, dst=path, remove=False)
 82
 83    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'."
 84
 85    return data_dir
 86
 87
 88def get_pyropia_paths(
 89    path: Union[os.PathLike, str], strain: Optional[Literal["sansha-5", "wo115-2"]] = None, download: bool = False,
 90) -> Tuple[List[str], List[str]]:
 91    """Get paths to the Pyropia data.
 92
 93    Args:
 94        path: Filepath to a folder where the data is downloaded for further processing.
 95        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
 96        download: Whether to download the data if it is not present.
 97
 98    Returns:
 99        List of filepaths for the image data.
100        List of filepaths for the label data.
101    """
102    if strain is not None and strain not in STRAINS:
103        raise ValueError(f"'{strain}' is not a valid strain. Choose one of {list(STRAINS)}.")
104
105    data_dir = get_pyropia_data(path, download)
106    label_dir = os.path.join(path, "labels")
107    os.makedirs(label_dir, exist_ok=True)
108
109    plants = [i for name, ids in STRAINS.items() if strain in (None, name) for i in ids]
110
111    raw_paths, label_paths = [], []
112    for plant in plants:
113        json_paths = natsorted(glob(os.path.join(data_dir, f"plant_{plant}", "json", "*.json")))
114        for json_path in json_paths:
115            stem = os.path.splitext(os.path.basename(json_path))[0]
116            raw_path = glob(os.path.join(data_dir, f"plant_{plant}", "img", f"{stem}.tif*"))
117            assert len(raw_path) == 1, f"Expected one image for '{json_path}', found {len(raw_path)}."
118
119            label_path = os.path.join(label_dir, f"plant_{plant}_{stem}.tif")
120            _rasterize_annotation(json_path, label_path)
121
122            raw_paths.append(raw_path[0])
123            label_paths.append(label_path)
124
125    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
126
127    return raw_paths, label_paths
128
129
130def get_pyropia_dataset(
131    path: Union[os.PathLike, str],
132    patch_shape: Tuple[int, int],
133    strain: Optional[Literal["sansha-5", "wo115-2"]] = None,
134    resize_inputs: bool = False,
135    download: bool = False,
136    **kwargs
137) -> Dataset:
138    """Get the Pyropia dataset for cell instance segmentation in plant microscopy images.
139
140    Args:
141        path: Filepath to a folder where the data is downloaded for further processing.
142        patch_shape: The patch shape to use for training.
143        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
144        resize_inputs: Whether to resize the inputs to the patch shape.
145        download: Whether to download the data if it is not present.
146        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
147
148    Returns:
149        The segmentation dataset.
150    """
151    raw_paths, label_paths = get_pyropia_paths(path, strain, download)
152
153    if resize_inputs:
154        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
155        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
156            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
157        )
158
159    return torch_em.default_segmentation_dataset(
160        raw_paths=raw_paths,
161        raw_key=None,
162        label_paths=label_paths,
163        label_key=None,
164        is_seg_dataset=False,
165        patch_shape=patch_shape,
166        **kwargs
167    )
168
169
170def get_pyropia_loader(
171    path: Union[os.PathLike, str],
172    batch_size: int,
173    patch_shape: Tuple[int, int],
174    strain: Optional[Literal["sansha-5", "wo115-2"]] = None,
175    resize_inputs: bool = False,
176    download: bool = False,
177    **kwargs
178) -> DataLoader:
179    """Get the Pyropia dataloader for cell instance segmentation in plant microscopy images.
180
181    Args:
182        path: Filepath to a folder where the data is downloaded for further processing.
183        batch_size: The batch size for training.
184        patch_shape: The patch shape to use for training.
185        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
186        resize_inputs: Whether to resize the inputs to the patch shape.
187        download: Whether to download the data if it is not present.
188        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
189
190    Returns:
191        The DataLoader.
192    """
193    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
194    dataset = get_pyropia_dataset(path, patch_shape, strain, resize_inputs, download, **ds_kwargs)
195    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/19571835/files/data.zip'
CHECKSUM = '55aa506e6fe6a68582f85ef5ac084a4107c686ae231a89ee9c3ae08f35558ecb'
STRAINS = {'sansha-5': range(1, 9), 'wo115-2': range(9, 17)}
def get_pyropia_data(path: Union[os.PathLike, str], download: bool = False) -> str:
64def get_pyropia_data(path: Union[os.PathLike, str], download: bool = False) -> str:
65    """Download the Pyropia dataset.
66
67    Args:
68        path: Filepath to a folder where the data is downloaded for further processing.
69        download: Whether to download the data if it is not present.
70
71    Returns:
72        Filepath where the data is downloaded.
73    """
74    data_dir = os.path.join(path, "data")
75    if os.path.exists(data_dir):
76        return data_dir
77
78    os.makedirs(path, exist_ok=True)
79
80    zip_path = os.path.join(path, "pyropia.zip")
81    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
82    util.unzip(zip_path=zip_path, dst=path, remove=False)
83
84    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'."
85
86    return data_dir

Download the Pyropia dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_pyropia_paths( path: Union[os.PathLike, str], strain: Optional[Literal['sansha-5', 'wo115-2']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 89def get_pyropia_paths(
 90    path: Union[os.PathLike, str], strain: Optional[Literal["sansha-5", "wo115-2"]] = None, download: bool = False,
 91) -> Tuple[List[str], List[str]]:
 92    """Get paths to the Pyropia data.
 93
 94    Args:
 95        path: Filepath to a folder where the data is downloaded for further processing.
 96        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
 97        download: Whether to download the data if it is not present.
 98
 99    Returns:
100        List of filepaths for the image data.
101        List of filepaths for the label data.
102    """
103    if strain is not None and strain not in STRAINS:
104        raise ValueError(f"'{strain}' is not a valid strain. Choose one of {list(STRAINS)}.")
105
106    data_dir = get_pyropia_data(path, download)
107    label_dir = os.path.join(path, "labels")
108    os.makedirs(label_dir, exist_ok=True)
109
110    plants = [i for name, ids in STRAINS.items() if strain in (None, name) for i in ids]
111
112    raw_paths, label_paths = [], []
113    for plant in plants:
114        json_paths = natsorted(glob(os.path.join(data_dir, f"plant_{plant}", "json", "*.json")))
115        for json_path in json_paths:
116            stem = os.path.splitext(os.path.basename(json_path))[0]
117            raw_path = glob(os.path.join(data_dir, f"plant_{plant}", "img", f"{stem}.tif*"))
118            assert len(raw_path) == 1, f"Expected one image for '{json_path}', found {len(raw_path)}."
119
120            label_path = os.path.join(label_dir, f"plant_{plant}_{stem}.tif")
121            _rasterize_annotation(json_path, label_path)
122
123            raw_paths.append(raw_path[0])
124            label_paths.append(label_path)
125
126    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
127
128    return raw_paths, label_paths

Get paths to the Pyropia data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_pyropia_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], strain: Optional[Literal['sansha-5', 'wo115-2']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
131def get_pyropia_dataset(
132    path: Union[os.PathLike, str],
133    patch_shape: Tuple[int, int],
134    strain: Optional[Literal["sansha-5", "wo115-2"]] = None,
135    resize_inputs: bool = False,
136    download: bool = False,
137    **kwargs
138) -> Dataset:
139    """Get the Pyropia dataset for cell instance segmentation in plant microscopy images.
140
141    Args:
142        path: Filepath to a folder where the data is downloaded for further processing.
143        patch_shape: The patch shape to use for training.
144        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
145        resize_inputs: Whether to resize the inputs to the patch shape.
146        download: Whether to download the data if it is not present.
147        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
148
149    Returns:
150        The segmentation dataset.
151    """
152    raw_paths, label_paths = get_pyropia_paths(path, strain, download)
153
154    if resize_inputs:
155        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
156        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
157            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
158        )
159
160    return torch_em.default_segmentation_dataset(
161        raw_paths=raw_paths,
162        raw_key=None,
163        label_paths=label_paths,
164        label_key=None,
165        is_seg_dataset=False,
166        patch_shape=patch_shape,
167        **kwargs
168    )

Get the Pyropia dataset for cell instance segmentation in plant microscopy images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pyropia_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], strain: Optional[Literal['sansha-5', 'wo115-2']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
171def get_pyropia_loader(
172    path: Union[os.PathLike, str],
173    batch_size: int,
174    patch_shape: Tuple[int, int],
175    strain: Optional[Literal["sansha-5", "wo115-2"]] = None,
176    resize_inputs: bool = False,
177    download: bool = False,
178    **kwargs
179) -> DataLoader:
180    """Get the Pyropia dataloader for cell instance segmentation in plant microscopy images.
181
182    Args:
183        path: Filepath to a folder where the data is downloaded for further processing.
184        batch_size: The batch size for training.
185        patch_shape: The patch shape to use for training.
186        strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
187        resize_inputs: Whether to resize the inputs to the patch shape.
188        download: Whether to download the data if it is not present.
189        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
190
191    Returns:
192        The DataLoader.
193    """
194    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
195    dataset = get_pyropia_dataset(path, patch_shape, strain, resize_inputs, download, **ds_kwargs)
196    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Pyropia dataloader for cell instance segmentation in plant microscopy images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • strain: The choice of strain. Either 'sansha-5' or 'wo115-2'. By default both strains are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.