torch_em.data.datasets.light_microscopy.deepfucci

The DeepFUCCI dataset contains annotations for nucleus instance segmentation and cell cycle classification in multiplexed FUCCI (fluorescent ubiquitination-based cell cycle indicator) fluorescence microscopy images.

The training data of the DeepFUCCI publication (the 'fucciphase' Zenodo record) consists of 513 images with 3 channels (float values, normalized per image and not in [0, 1]) and sizes between 256x256 and 500x500 pixels, from several cell lines and acquisitions (e.g. HaCaT cells at 100x, a scratch assay, 20x and 40x recordings). Each image has an instance mask following the StarDist convention (0 is background, every nucleus has its own id) and a json file that maps every instance id to one of the 3 cell cycle classes 1, 2 and 3. The authors define the split of the images ('training': 436 and 'validation': 77 images) in 'dataset_split.json'. The order of the 3 channels and the meaning of the class ids are not documented in the record, so they are not interpreted by this module.

The record also contains a small independent test set (data_set_HT1080.zip), custom trained StarDist, InstanSeg and Cellpose-SAM models and analysis data, which are not used here.

The data is located at https://doi.org/10.5281/zenodo.19671003, released under a CC-BY-4.0 license. Please cite the corresponding Zenodo record if you use this dataset in your research.

  1"""The DeepFUCCI dataset contains annotations for nucleus instance segmentation and cell cycle classification
  2in multiplexed FUCCI (fluorescent ubiquitination-based cell cycle indicator) fluorescence microscopy images.
  3
  4The training data of the DeepFUCCI publication (the 'fucciphase' Zenodo record) consists of 513 images with 3 channels
  5(float values, normalized per image and not in [0, 1]) and sizes between 256x256 and 500x500 pixels, from several
  6cell lines and acquisitions (e.g. HaCaT cells at 100x, a scratch assay, 20x and 40x recordings). Each image has an
  7instance mask following the StarDist convention (0 is background, every nucleus has its own id) and a json file that
  8maps every instance id to one of the 3 cell cycle classes 1, 2 and 3. The authors define the split of the images
  9('training': 436 and 'validation': 77 images) in 'dataset_split.json'. The order of the 3 channels and the meaning of
 10the class ids are not documented in the record, so they are not interpreted by this module.
 11
 12The record also contains a small independent test set (data_set_HT1080.zip), custom trained StarDist, InstanSeg and
 13Cellpose-SAM models and analysis data, which are not used here.
 14
 15The data is located at https://doi.org/10.5281/zenodo.19671003, released under a CC-BY-4.0 license.
 16Please cite the corresponding Zenodo record if you use this dataset in your research.
 17"""
 18
 19import os
 20import json
 21from natsort import natsorted
 22from typing import Union, Tuple, Literal, List
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URL = "https://zenodo.org/api/records/19671003/files/training_data.zip/content"
 32CHECKSUM = "03be8d4caf38cd7780ea2043a32697ef63defd25468f4953a8844890545f668b"
 33
 34SPLITS = {"train": "training", "val": "validation"}
 35
 36
 37def get_deepfucci_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 38    """Download the DeepFUCCI training data.
 39
 40    Args:
 41        path: Filepath to a folder where the downloaded data will be saved.
 42        download: Whether to download the data if it is not present.
 43
 44    Returns:
 45        The filepath to the extracted data.
 46    """
 47    data_dir = os.path.join(path, "training_data")
 48    if os.path.exists(os.path.join(data_dir, "dataset_split.json")):
 49        return data_dir
 50
 51    os.makedirs(path, exist_ok=True)
 52    zip_path = os.path.join(path, "training_data.zip")
 53    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 54    util.unzip(zip_path=zip_path, dst=path, remove=False)
 55
 56    assert os.path.exists(data_dir), f"The extraction of the archive did not create '{data_dir}'."
 57    return data_dir
 58
 59
 60def get_deepfucci_paths(
 61    path: Union[os.PathLike, str], split: Literal["train", "val"] = "train", download: bool = False,
 62) -> Tuple[List[str], List[str]]:
 63    """Get paths to the DeepFUCCI training data.
 64
 65    Args:
 66        path: Filepath to a folder where the downloaded data will be saved.
 67        split: The choice of data split. Either 'train' or 'val'.
 68        download: Whether to download the data if it is not present.
 69
 70    Returns:
 71        List of filepaths for the image data.
 72        List of filepaths for the label data.
 73    """
 74    if split not in SPLITS:
 75        raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.")
 76
 77    data_dir = get_deepfucci_data(path, download)
 78    with open(os.path.join(data_dir, "dataset_split.json")) as f:
 79        names = natsorted(json.load(f)[SPLITS[split]])
 80
 81    raw_paths = [os.path.join(data_dir, "images", name) for name in names]
 82    label_paths = [os.path.join(data_dir, "masks", name) for name in names]
 83
 84    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 85    assert all(os.path.exists(p) for p in raw_paths + label_paths)
 86
 87    return raw_paths, label_paths
 88
 89
 90def get_deepfucci_dataset(
 91    path: Union[os.PathLike, str],
 92    patch_shape: Tuple[int, int],
 93    split: Literal["train", "val"] = "train",
 94    download: bool = False,
 95    **kwargs
 96) -> Dataset:
 97    """Get the DeepFUCCI dataset for nucleus instance segmentation in multiplexed FUCCI microscopy images.
 98
 99    Args:
100        path: Filepath to a folder where the downloaded data will be saved.
101        patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
102        split: The choice of data split. Either 'train' or 'val'.
103        download: Whether to download the data if it is not present.
104        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
105
106    Returns:
107        The segmentation dataset.
108    """
109    raw_paths, label_paths = get_deepfucci_paths(path, split, download)
110
111    return torch_em.default_segmentation_dataset(
112        raw_paths=raw_paths,
113        raw_key=None,
114        label_paths=label_paths,
115        label_key=None,
116        patch_shape=patch_shape,
117        is_seg_dataset=False,
118        **kwargs
119    )
120
121
122def get_deepfucci_loader(
123    path: Union[os.PathLike, str],
124    batch_size: int,
125    patch_shape: Tuple[int, int],
126    split: Literal["train", "val"] = "train",
127    download: bool = False,
128    **kwargs
129) -> DataLoader:
130    """Get the DeepFUCCI dataloader for nucleus instance segmentation in multiplexed FUCCI microscopy images.
131
132    Args:
133        path: Filepath to a folder where the downloaded data will be saved.
134        batch_size: The batch size for training.
135        patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
136        split: The choice of data split. Either 'train' or 'val'.
137        download: Whether to download the data if it is not present.
138        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
139
140    Returns:
141        The DataLoader.
142    """
143    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
144    dataset = get_deepfucci_dataset(path, patch_shape, split, download, **ds_kwargs)
145    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/api/records/19671003/files/training_data.zip/content'
CHECKSUM = '03be8d4caf38cd7780ea2043a32697ef63defd25468f4953a8844890545f668b'
SPLITS = {'train': 'training', 'val': 'validation'}
def get_deepfucci_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38def get_deepfucci_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39    """Download the DeepFUCCI training data.
40
41    Args:
42        path: Filepath to a folder where the downloaded data will be saved.
43        download: Whether to download the data if it is not present.
44
45    Returns:
46        The filepath to the extracted data.
47    """
48    data_dir = os.path.join(path, "training_data")
49    if os.path.exists(os.path.join(data_dir, "dataset_split.json")):
50        return data_dir
51
52    os.makedirs(path, exist_ok=True)
53    zip_path = os.path.join(path, "training_data.zip")
54    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
55    util.unzip(zip_path=zip_path, dst=path, remove=False)
56
57    assert os.path.exists(data_dir), f"The extraction of the archive did not create '{data_dir}'."
58    return data_dir

Download the DeepFUCCI training data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • download: Whether to download the data if it is not present.
Returns:

The filepath to the extracted data.

def get_deepfucci_paths( path: Union[os.PathLike, str], split: Literal['train', 'val'] = 'train', download: bool = False) -> Tuple[List[str], List[str]]:
61def get_deepfucci_paths(
62    path: Union[os.PathLike, str], split: Literal["train", "val"] = "train", download: bool = False,
63) -> Tuple[List[str], List[str]]:
64    """Get paths to the DeepFUCCI training data.
65
66    Args:
67        path: Filepath to a folder where the downloaded data will be saved.
68        split: The choice of data split. Either 'train' or 'val'.
69        download: Whether to download the data if it is not present.
70
71    Returns:
72        List of filepaths for the image data.
73        List of filepaths for the label data.
74    """
75    if split not in SPLITS:
76        raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.")
77
78    data_dir = get_deepfucci_data(path, download)
79    with open(os.path.join(data_dir, "dataset_split.json")) as f:
80        names = natsorted(json.load(f)[SPLITS[split]])
81
82    raw_paths = [os.path.join(data_dir, "images", name) for name in names]
83    label_paths = [os.path.join(data_dir, "masks", name) for name in names]
84
85    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
86    assert all(os.path.exists(p) for p in raw_paths + label_paths)
87
88    return raw_paths, label_paths

Get paths to the DeepFUCCI training data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • split: The choice of data split. Either 'train' or 'val'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_deepfucci_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val'] = 'train', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 91def get_deepfucci_dataset(
 92    path: Union[os.PathLike, str],
 93    patch_shape: Tuple[int, int],
 94    split: Literal["train", "val"] = "train",
 95    download: bool = False,
 96    **kwargs
 97) -> Dataset:
 98    """Get the DeepFUCCI dataset for nucleus instance segmentation in multiplexed FUCCI microscopy images.
 99
100    Args:
101        path: Filepath to a folder where the downloaded data will be saved.
102        patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
103        split: The choice of data split. Either 'train' or 'val'.
104        download: Whether to download the data if it is not present.
105        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
106
107    Returns:
108        The segmentation dataset.
109    """
110    raw_paths, label_paths = get_deepfucci_paths(path, split, download)
111
112    return torch_em.default_segmentation_dataset(
113        raw_paths=raw_paths,
114        raw_key=None,
115        label_paths=label_paths,
116        label_key=None,
117        patch_shape=patch_shape,
118        is_seg_dataset=False,
119        **kwargs
120    )

Get the DeepFUCCI dataset for nucleus instance segmentation in multiplexed FUCCI microscopy images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
  • split: The choice of data split. Either 'train' or 'val'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_deepfucci_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val'] = 'train', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
123def get_deepfucci_loader(
124    path: Union[os.PathLike, str],
125    batch_size: int,
126    patch_shape: Tuple[int, int],
127    split: Literal["train", "val"] = "train",
128    download: bool = False,
129    **kwargs
130) -> DataLoader:
131    """Get the DeepFUCCI dataloader for nucleus instance segmentation in multiplexed FUCCI microscopy images.
132
133    Args:
134        path: Filepath to a folder where the downloaded data will be saved.
135        batch_size: The batch size for training.
136        patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
137        split: The choice of data split. Either 'train' or 'val'.
138        download: Whether to download the data if it is not present.
139        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
140
141    Returns:
142        The DataLoader.
143    """
144    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
145    dataset = get_deepfucci_dataset(path, patch_shape, split, download, **ds_kwargs)
146    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DeepFUCCI dataloader for nucleus instance segmentation in multiplexed FUCCI microscopy images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training. The smallest images have a size of 256x256 pixels.
  • split: The choice of data split. Either 'train' or 'val'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.