torch_em.data.datasets.light_microscopy.goblet_cell
The GCS dataset contains annotations for goblet cell instance segmentation in microscopy images of the human conjunctiva.
The dataset consists of 24 images (2048x1536, RGB) with 65,108 manually annotated single cells. It also provides a patched version of the same images with 1,152 patches (256x256) and 75,597 instances, where cells that cross patch borders are counted in every patch that contains them. The masks store one integer id per cell, which is used directly as the instance label (touching cells are not merged).
The data is located at https://doi.org/10.5281/zenodo.18517381 (latest version: https://zenodo.org/records/18642562), released under a CC-BY-4.0 license. The archive also contains YOLO and SAM2 annotation formats, which are not used.
The official splits are used for the unpatched images and for the patched images split by source image ('img_level'). The second official patched split ('random') puts patches of the same source image into train and test and is therefore not exposed.
This dataset is from the publication https://doi.org/10.1038/s41597-026-07309-w. Please cite it if you use this dataset for your research.
1"""The GCS dataset contains annotations for goblet cell instance segmentation in 2microscopy images of the human conjunctiva. 3 4The dataset consists of 24 images (2048x1536, RGB) with 65,108 manually annotated single cells. It also 5provides a patched version of the same images with 1,152 patches (256x256) and 75,597 instances, where cells 6that cross patch borders are counted in every patch that contains them. The masks store one integer id per cell, 7which is used directly as the instance label (touching cells are not merged). 8 9The data is located at https://doi.org/10.5281/zenodo.18517381 (latest version: https://zenodo.org/records/18642562), 10released under a CC-BY-4.0 license. The archive also contains YOLO and SAM2 annotation formats, which are not used. 11 12The official splits are used for the unpatched images and for the patched images split by source image ('img_level'). 13The second official patched split ('random') puts patches of the same source image into train and test and is 14therefore not exposed. 15 16This dataset is from the publication https://doi.org/10.1038/s41597-026-07309-w. 17Please cite it if you use this dataset for your research. 18""" 19 20import os 21from glob import glob 22from natsort import natsorted 23from typing import Union, Tuple, Literal, List 24 25from torch.utils.data import Dataset, DataLoader 26 27import torch_em 28 29from .. import util 30 31 32URL = "https://zenodo.org/api/records/18642562/files/GCSdataV1.1.zip/content" 33CHECKSUM = "e0c59c5d6c281b3f793b934e605c16ca80d8d75f59da3f2579ca6da9c0c0a973" 34 35SPLITS = ["train", "test", "all"] 36 37 38def get_goblet_cell_data(path: Union[os.PathLike, str], download: bool = False) -> str: 39 """Download the GCS dataset. 40 41 Args: 42 path: Filepath to a folder where the data is downloaded for further processing. 43 download: Whether to download the data if it is not present. 44 45 Returns: 46 Filepath to the folder with the original images and masks. 47 """ 48 data_dir = os.path.join(path, "GCSdataV1.1", "original_data") 49 if os.path.exists(data_dir): 50 return data_dir 51 52 os.makedirs(path, exist_ok=True) 53 54 zip_path = os.path.join(path, "GCSdataV1.1.zip") 55 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 56 util.unzip(zip_path=zip_path, dst=path, remove=False) 57 58 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'." 59 60 return data_dir 61 62 63def get_goblet_cell_paths( 64 path: Union[os.PathLike, str], 65 split: Literal["train", "test", "all"] = "all", 66 patched: bool = False, 67 download: bool = False, 68) -> Tuple[List[str], List[str]]: 69 """Get paths to the GCS data. 70 71 Args: 72 path: Filepath to a folder where the data is downloaded for further processing. 73 split: The choice of data split. One of 'train', 'test' or 'all'. 74 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 List of filepaths for the image data. 79 List of filepaths for the label data. 80 """ 81 if split not in SPLITS: 82 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 83 84 data_dir = get_goblet_cell_data(path, download) 85 86 version = "patched" if patched else "unpatched" 87 if split == "all": 88 split_dir = os.path.join(data_dir, "complete", version) 89 else: 90 split_dir = os.path.join(data_dir, "train_test_split", "patched_img_level" if patched else "unpatched", split) 91 92 raw_paths = natsorted(glob(os.path.join(split_dir, "images", "*"))) 93 label_paths = [ 94 os.path.join(split_dir, "masks", f"{os.path.splitext(os.path.basename(p))[0]}.png") for p in raw_paths 95 ] 96 97 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 98 assert all(os.path.exists(p) for p in label_paths) 99 100 return raw_paths, label_paths 101 102 103def get_goblet_cell_dataset( 104 path: Union[os.PathLike, str], 105 patch_shape: Tuple[int, int], 106 split: Literal["train", "test", "all"] = "all", 107 patched: bool = False, 108 resize_inputs: bool = False, 109 download: bool = False, 110 **kwargs 111) -> Dataset: 112 """Get the GCS dataset for goblet cell instance segmentation. 113 114 Args: 115 path: Filepath to a folder where the data is downloaded for further processing. 116 patch_shape: The patch shape to use for training. 117 split: The choice of data split. One of 'train', 'test' or 'all'. 118 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 119 resize_inputs: Whether to resize the inputs to the patch shape. 120 download: Whether to download the data if it is not present. 121 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 122 123 Returns: 124 The segmentation dataset. 125 """ 126 raw_paths, label_paths = get_goblet_cell_paths(path, split, patched, download) 127 128 if resize_inputs: 129 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 130 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 131 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 132 ) 133 134 return torch_em.default_segmentation_dataset( 135 raw_paths=raw_paths, 136 raw_key=None, 137 label_paths=label_paths, 138 label_key=None, 139 patch_shape=patch_shape, 140 is_seg_dataset=False, 141 **kwargs 142 ) 143 144 145def get_goblet_cell_loader( 146 path: Union[os.PathLike, str], 147 batch_size: int, 148 patch_shape: Tuple[int, int], 149 split: Literal["train", "test", "all"] = "all", 150 patched: bool = False, 151 resize_inputs: bool = False, 152 download: bool = False, 153 **kwargs 154) -> DataLoader: 155 """Get the GCS dataloader for goblet cell instance segmentation. 156 157 Args: 158 path: Filepath to a folder where the data is downloaded for further processing. 159 batch_size: The batch size for training. 160 patch_shape: The patch shape to use for training. 161 split: The choice of data split. One of 'train', 'test' or 'all'. 162 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 163 resize_inputs: Whether to resize the inputs to the patch shape. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 166 167 Returns: 168 The DataLoader. 169 """ 170 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 171 dataset = get_goblet_cell_dataset(path, patch_shape, split, patched, resize_inputs, download, **ds_kwargs) 172 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
39def get_goblet_cell_data(path: Union[os.PathLike, str], download: bool = False) -> str: 40 """Download the GCS dataset. 41 42 Args: 43 path: Filepath to a folder where the data is downloaded for further processing. 44 download: Whether to download the data if it is not present. 45 46 Returns: 47 Filepath to the folder with the original images and masks. 48 """ 49 data_dir = os.path.join(path, "GCSdataV1.1", "original_data") 50 if os.path.exists(data_dir): 51 return data_dir 52 53 os.makedirs(path, exist_ok=True) 54 55 zip_path = os.path.join(path, "GCSdataV1.1.zip") 56 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 57 util.unzip(zip_path=zip_path, dst=path, remove=False) 58 59 assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'." 60 61 return data_dir
Download the GCS dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder with the original images and masks.
64def get_goblet_cell_paths( 65 path: Union[os.PathLike, str], 66 split: Literal["train", "test", "all"] = "all", 67 patched: bool = False, 68 download: bool = False, 69) -> Tuple[List[str], List[str]]: 70 """Get paths to the GCS data. 71 72 Args: 73 path: Filepath to a folder where the data is downloaded for further processing. 74 split: The choice of data split. One of 'train', 'test' or 'all'. 75 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 76 download: Whether to download the data if it is not present. 77 78 Returns: 79 List of filepaths for the image data. 80 List of filepaths for the label data. 81 """ 82 if split not in SPLITS: 83 raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.") 84 85 data_dir = get_goblet_cell_data(path, download) 86 87 version = "patched" if patched else "unpatched" 88 if split == "all": 89 split_dir = os.path.join(data_dir, "complete", version) 90 else: 91 split_dir = os.path.join(data_dir, "train_test_split", "patched_img_level" if patched else "unpatched", split) 92 93 raw_paths = natsorted(glob(os.path.join(split_dir, "images", "*"))) 94 label_paths = [ 95 os.path.join(split_dir, "masks", f"{os.path.splitext(os.path.basename(p))[0]}.png") for p in raw_paths 96 ] 97 98 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 99 assert all(os.path.exists(p) for p in label_paths) 100 101 return raw_paths, label_paths
Get paths to the GCS data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. One of 'train', 'test' or 'all'.
- patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
104def get_goblet_cell_dataset( 105 path: Union[os.PathLike, str], 106 patch_shape: Tuple[int, int], 107 split: Literal["train", "test", "all"] = "all", 108 patched: bool = False, 109 resize_inputs: bool = False, 110 download: bool = False, 111 **kwargs 112) -> Dataset: 113 """Get the GCS dataset for goblet cell instance segmentation. 114 115 Args: 116 path: Filepath to a folder where the data is downloaded for further processing. 117 patch_shape: The patch shape to use for training. 118 split: The choice of data split. One of 'train', 'test' or 'all'. 119 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 120 resize_inputs: Whether to resize the inputs to the patch shape. 121 download: Whether to download the data if it is not present. 122 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 123 124 Returns: 125 The segmentation dataset. 126 """ 127 raw_paths, label_paths = get_goblet_cell_paths(path, split, patched, download) 128 129 if resize_inputs: 130 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 131 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 132 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 133 ) 134 135 return torch_em.default_segmentation_dataset( 136 raw_paths=raw_paths, 137 raw_key=None, 138 label_paths=label_paths, 139 label_key=None, 140 patch_shape=patch_shape, 141 is_seg_dataset=False, 142 **kwargs 143 )
Get the GCS dataset for goblet cell instance segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. One of 'train', 'test' or 'all'.
- patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
146def get_goblet_cell_loader( 147 path: Union[os.PathLike, str], 148 batch_size: int, 149 patch_shape: Tuple[int, int], 150 split: Literal["train", "test", "all"] = "all", 151 patched: bool = False, 152 resize_inputs: bool = False, 153 download: bool = False, 154 **kwargs 155) -> DataLoader: 156 """Get the GCS dataloader for goblet cell instance segmentation. 157 158 Args: 159 path: Filepath to a folder where the data is downloaded for further processing. 160 batch_size: The batch size for training. 161 patch_shape: The patch shape to use for training. 162 split: The choice of data split. One of 'train', 'test' or 'all'. 163 patched: Whether to use the patched images (256x256) instead of the full images (2048x1536). 164 resize_inputs: Whether to resize the inputs to the patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 167 168 Returns: 169 The DataLoader. 170 """ 171 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 172 dataset = get_goblet_cell_dataset(path, patch_shape, split, patched, resize_inputs, download, **ds_kwargs) 173 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the GCS dataloader for goblet cell instance segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split. One of 'train', 'test' or 'all'.
- patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.