torch_em.data.datasets.light_microscopy.goblet_cell

The GCS dataset contains annotations for goblet cell instance segmentation in microscopy images of the human conjunctiva.

The dataset consists of 24 images (2048x1536, RGB) with 65,108 manually annotated single cells. It also provides a patched version of the same images with 1,152 patches (256x256) and 75,597 instances, where cells that cross patch borders are counted in every patch that contains them. The masks store one integer id per cell, which is used directly as the instance label (touching cells are not merged).

The data is located at https://doi.org/10.5281/zenodo.18517381 (latest version: https://zenodo.org/records/18642562), released under a CC-BY-4.0 license. The archive also contains YOLO and SAM2 annotation formats, which are not used.

The official splits are used for the unpatched images and for the patched images split by source image ('img_level'). The second official patched split ('random') puts patches of the same source image into train and test and is therefore not exposed.

This dataset is from the publication https://doi.org/10.1038/s41597-026-07309-w. Please cite it if you use this dataset for your research.

  1"""The GCS dataset contains annotations for goblet cell instance segmentation in
  2microscopy images of the human conjunctiva.
  3
  4The dataset consists of 24 images (2048x1536, RGB) with 65,108 manually annotated single cells. It also
  5provides a patched version of the same images with 1,152 patches (256x256) and 75,597 instances, where cells
  6that cross patch borders are counted in every patch that contains them. The masks store one integer id per cell,
  7which is used directly as the instance label (touching cells are not merged).
  8
  9The data is located at https://doi.org/10.5281/zenodo.18517381 (latest version: https://zenodo.org/records/18642562),
 10released under a CC-BY-4.0 license. The archive also contains YOLO and SAM2 annotation formats, which are not used.
 11
 12The official splits are used for the unpatched images and for the patched images split by source image ('img_level').
 13The second official patched split ('random') puts patches of the same source image into train and test and is
 14therefore not exposed.
 15
 16This dataset is from the publication https://doi.org/10.1038/s41597-026-07309-w.
 17Please cite it if you use this dataset for your research.
 18"""
 19
 20import os
 21from glob import glob
 22from natsort import natsorted
 23from typing import Union, Tuple, Literal, List
 24
 25from torch.utils.data import Dataset, DataLoader
 26
 27import torch_em
 28
 29from .. import util
 30
 31
 32URL = "https://zenodo.org/api/records/18642562/files/GCSdataV1.1.zip/content"
 33CHECKSUM = "e0c59c5d6c281b3f793b934e605c16ca80d8d75f59da3f2579ca6da9c0c0a973"
 34
 35SPLITS = ["train", "test", "all"]
 36
 37
 38def get_goblet_cell_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 39    """Download the GCS dataset.
 40
 41    Args:
 42        path: Filepath to a folder where the data is downloaded for further processing.
 43        download: Whether to download the data if it is not present.
 44
 45    Returns:
 46        Filepath to the folder with the original images and masks.
 47    """
 48    data_dir = os.path.join(path, "GCSdataV1.1", "original_data")
 49    if os.path.exists(data_dir):
 50        return data_dir
 51
 52    os.makedirs(path, exist_ok=True)
 53
 54    zip_path = os.path.join(path, "GCSdataV1.1.zip")
 55    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 56    util.unzip(zip_path=zip_path, dst=path, remove=False)
 57
 58    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'."
 59
 60    return data_dir
 61
 62
 63def get_goblet_cell_paths(
 64    path: Union[os.PathLike, str],
 65    split: Literal["train", "test", "all"] = "all",
 66    patched: bool = False,
 67    download: bool = False,
 68) -> Tuple[List[str], List[str]]:
 69    """Get paths to the GCS data.
 70
 71    Args:
 72        path: Filepath to a folder where the data is downloaded for further processing.
 73        split: The choice of data split. One of 'train', 'test' or 'all'.
 74        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
 75        download: Whether to download the data if it is not present.
 76
 77    Returns:
 78        List of filepaths for the image data.
 79        List of filepaths for the label data.
 80    """
 81    if split not in SPLITS:
 82        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
 83
 84    data_dir = get_goblet_cell_data(path, download)
 85
 86    version = "patched" if patched else "unpatched"
 87    if split == "all":
 88        split_dir = os.path.join(data_dir, "complete", version)
 89    else:
 90        split_dir = os.path.join(data_dir, "train_test_split", "patched_img_level" if patched else "unpatched", split)
 91
 92    raw_paths = natsorted(glob(os.path.join(split_dir, "images", "*")))
 93    label_paths = [
 94        os.path.join(split_dir, "masks", f"{os.path.splitext(os.path.basename(p))[0]}.png") for p in raw_paths
 95    ]
 96
 97    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 98    assert all(os.path.exists(p) for p in label_paths)
 99
100    return raw_paths, label_paths
101
102
103def get_goblet_cell_dataset(
104    path: Union[os.PathLike, str],
105    patch_shape: Tuple[int, int],
106    split: Literal["train", "test", "all"] = "all",
107    patched: bool = False,
108    resize_inputs: bool = False,
109    download: bool = False,
110    **kwargs
111) -> Dataset:
112    """Get the GCS dataset for goblet cell instance segmentation.
113
114    Args:
115        path: Filepath to a folder where the data is downloaded for further processing.
116        patch_shape: The patch shape to use for training.
117        split: The choice of data split. One of 'train', 'test' or 'all'.
118        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
119        resize_inputs: Whether to resize the inputs to the patch shape.
120        download: Whether to download the data if it is not present.
121        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
122
123    Returns:
124        The segmentation dataset.
125    """
126    raw_paths, label_paths = get_goblet_cell_paths(path, split, patched, download)
127
128    if resize_inputs:
129        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
130        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
131            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
132        )
133
134    return torch_em.default_segmentation_dataset(
135        raw_paths=raw_paths,
136        raw_key=None,
137        label_paths=label_paths,
138        label_key=None,
139        patch_shape=patch_shape,
140        is_seg_dataset=False,
141        **kwargs
142    )
143
144
145def get_goblet_cell_loader(
146    path: Union[os.PathLike, str],
147    batch_size: int,
148    patch_shape: Tuple[int, int],
149    split: Literal["train", "test", "all"] = "all",
150    patched: bool = False,
151    resize_inputs: bool = False,
152    download: bool = False,
153    **kwargs
154) -> DataLoader:
155    """Get the GCS dataloader for goblet cell instance segmentation.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        batch_size: The batch size for training.
160        patch_shape: The patch shape to use for training.
161        split: The choice of data split. One of 'train', 'test' or 'all'.
162        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
163        resize_inputs: Whether to resize the inputs to the patch shape.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_goblet_cell_dataset(path, patch_shape, split, patched, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/api/records/18642562/files/GCSdataV1.1.zip/content'
CHECKSUM = 'e0c59c5d6c281b3f793b934e605c16ca80d8d75f59da3f2579ca6da9c0c0a973'
SPLITS = ['train', 'test', 'all']
def get_goblet_cell_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39def get_goblet_cell_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40    """Download the GCS dataset.
41
42    Args:
43        path: Filepath to a folder where the data is downloaded for further processing.
44        download: Whether to download the data if it is not present.
45
46    Returns:
47        Filepath to the folder with the original images and masks.
48    """
49    data_dir = os.path.join(path, "GCSdataV1.1", "original_data")
50    if os.path.exists(data_dir):
51        return data_dir
52
53    os.makedirs(path, exist_ok=True)
54
55    zip_path = os.path.join(path, "GCSdataV1.1.zip")
56    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
57    util.unzip(zip_path=zip_path, dst=path, remove=False)
58
59    assert os.path.exists(data_dir), f"The extraction of the archive did not create the expected folder in '{path}'."
60
61    return data_dir

Download the GCS dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the original images and masks.

def get_goblet_cell_paths( path: Union[os.PathLike, str], split: Literal['train', 'test', 'all'] = 'all', patched: bool = False, download: bool = False) -> Tuple[List[str], List[str]]:
 64def get_goblet_cell_paths(
 65    path: Union[os.PathLike, str],
 66    split: Literal["train", "test", "all"] = "all",
 67    patched: bool = False,
 68    download: bool = False,
 69) -> Tuple[List[str], List[str]]:
 70    """Get paths to the GCS data.
 71
 72    Args:
 73        path: Filepath to a folder where the data is downloaded for further processing.
 74        split: The choice of data split. One of 'train', 'test' or 'all'.
 75        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
 76        download: Whether to download the data if it is not present.
 77
 78    Returns:
 79        List of filepaths for the image data.
 80        List of filepaths for the label data.
 81    """
 82    if split not in SPLITS:
 83        raise ValueError(f"'{split}' is not a valid split. Choose one of {SPLITS}.")
 84
 85    data_dir = get_goblet_cell_data(path, download)
 86
 87    version = "patched" if patched else "unpatched"
 88    if split == "all":
 89        split_dir = os.path.join(data_dir, "complete", version)
 90    else:
 91        split_dir = os.path.join(data_dir, "train_test_split", "patched_img_level" if patched else "unpatched", split)
 92
 93    raw_paths = natsorted(glob(os.path.join(split_dir, "images", "*")))
 94    label_paths = [
 95        os.path.join(split_dir, "masks", f"{os.path.splitext(os.path.basename(p))[0]}.png") for p in raw_paths
 96    ]
 97
 98    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 99    assert all(os.path.exists(p) for p in label_paths)
100
101    return raw_paths, label_paths

Get paths to the GCS data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. One of 'train', 'test' or 'all'.
  • patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_goblet_cell_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'test', 'all'] = 'all', patched: bool = False, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
104def get_goblet_cell_dataset(
105    path: Union[os.PathLike, str],
106    patch_shape: Tuple[int, int],
107    split: Literal["train", "test", "all"] = "all",
108    patched: bool = False,
109    resize_inputs: bool = False,
110    download: bool = False,
111    **kwargs
112) -> Dataset:
113    """Get the GCS dataset for goblet cell instance segmentation.
114
115    Args:
116        path: Filepath to a folder where the data is downloaded for further processing.
117        patch_shape: The patch shape to use for training.
118        split: The choice of data split. One of 'train', 'test' or 'all'.
119        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
120        resize_inputs: Whether to resize the inputs to the patch shape.
121        download: Whether to download the data if it is not present.
122        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
123
124    Returns:
125        The segmentation dataset.
126    """
127    raw_paths, label_paths = get_goblet_cell_paths(path, split, patched, download)
128
129    if resize_inputs:
130        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
131        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
132            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
133        )
134
135    return torch_em.default_segmentation_dataset(
136        raw_paths=raw_paths,
137        raw_key=None,
138        label_paths=label_paths,
139        label_key=None,
140        patch_shape=patch_shape,
141        is_seg_dataset=False,
142        **kwargs
143    )

Get the GCS dataset for goblet cell instance segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. One of 'train', 'test' or 'all'.
  • patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_goblet_cell_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'test', 'all'] = 'all', patched: bool = False, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
146def get_goblet_cell_loader(
147    path: Union[os.PathLike, str],
148    batch_size: int,
149    patch_shape: Tuple[int, int],
150    split: Literal["train", "test", "all"] = "all",
151    patched: bool = False,
152    resize_inputs: bool = False,
153    download: bool = False,
154    **kwargs
155) -> DataLoader:
156    """Get the GCS dataloader for goblet cell instance segmentation.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        batch_size: The batch size for training.
161        patch_shape: The patch shape to use for training.
162        split: The choice of data split. One of 'train', 'test' or 'all'.
163        patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
164        resize_inputs: Whether to resize the inputs to the patch shape.
165        download: Whether to download the data if it is not present.
166        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
167
168    Returns:
169        The DataLoader.
170    """
171    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
172    dataset = get_goblet_cell_dataset(path, patch_shape, split, patched, resize_inputs, download, **ds_kwargs)
173    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the GCS dataloader for goblet cell instance segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. One of 'train', 'test' or 'all'.
  • patched: Whether to use the patched images (256x256) instead of the full images (2048x1536).
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.