torch_em.data.datasets.light_microscopy.mif_tonsil

The Multiplex IF Tonsil dataset contains annotations for nucleus and whole-cell instance segmentation in multiplex immunofluorescence images of human tonsil tissue.

The dataset consists of 10 image regions (100x100 pixels, 7 channels: CD21, CD23, CD20, CD4, CK, CD8 and DAPI), each with a manually annotated nuclear mask and a whole-cell mask ('.tif', '_GT_nuclei.tif' and '_GT_cells.tif'). The masks are instance labels, where the nucleus and the cell of an object share the same id.

The dataset is located at https://doi.org/10.5281/zenodo.22107836, released under a CC-BY-4.0 license. Please cite the corresponding Zenodo record if you use this dataset in your research.

  1"""The Multiplex IF Tonsil dataset contains annotations for nucleus and whole-cell instance segmentation
  2in multiplex immunofluorescence images of human tonsil tissue.
  3
  4The dataset consists of 10 image regions (100x100 pixels, 7 channels: CD21, CD23, CD20, CD4, CK, CD8 and DAPI),
  5each with a manually annotated nuclear mask and a whole-cell mask ('<name>.tif', '<name>_GT_nuclei.tif' and
  6'<name>_GT_cells.tif'). The masks are instance labels, where the nucleus and the cell of an object share the same id.
  7
  8The dataset is located at https://doi.org/10.5281/zenodo.22107836, released under a CC-BY-4.0 license.
  9Please cite the corresponding Zenodo record if you use this dataset in your research.
 10"""
 11
 12import os
 13from glob import glob
 14from natsort import natsorted
 15from typing import Union, Tuple, Literal, List
 16
 17from torch.utils.data import Dataset, DataLoader
 18
 19import torch_em
 20
 21from .. import util
 22
 23
 24BASE_URL = "https://zenodo.org/api/records/22107836/files"
 25
 26CHECKSUMS = {
 27    "23B10981_1_5_GT_cells.tif": "e14a640cb0a9d0bfa05c6f4b8418a0dee498b372afdc0f557e37c9f22822a764",
 28    "23B10981_1_5_GT_nuclei.tif": "9473b9406afff196ac932da6692a73748fbb9c20f4681ac9708723ea6c725d4e",
 29    "23B10981_1_5.tif": "9f04251bd4d13210343fcc36b41d3ffb59b248897f80a2018e9568d81c4f050f",
 30    "23B10981_9_15_GT_cells.tif": "5a9c323c2a181e76318e9d2b07b91e815a7443b9b16318561e538d72858e5b95",
 31    "23B10981_9_15_GT_nuclei.tif": "06afc949650b5be542ded08a3e6daae5e3c39dc91db0c7d6f7059b6c00d53dfd",
 32    "23B10981_9_15.tif": "f68d759e7ef734f30077333380132fb0bf82ed7c92efb96c581c9b2e383a1f0f",
 33    "24B2274_43_47_GT_cells.tif": "83e8b0810221d5fa3e208d2f8a27408c93bf52120a746d7a3840dd62945d2584",
 34    "24B2274_43_47_GT_nuclei.tif": "8c0871b27dc5bb2a3ab1443cb12011f7ee5f3f7dffb7e1d67b07499501a41d5f",
 35    "24B2274_43_47.tif": "b9b3a0f37089819c7604294fe97904660ff5097454feb5d7bea5c3a1fde25abc",
 36    "24B2274_45_30_GT_cells.tif": "6de91c905770ac92c8b115a251657e42fe17ca647b7fb0572f5b3e12e8926cd4",
 37    "24B2274_45_30_GT_nuclei.tif": "4e43d9c4e27be42bac4b3ab8c29adff6d1d6879887fcccff5ce74e222b635ac2",
 38    "24B2274_45_30.tif": "9f98bdd7d158221ba84d5c5b3a3c22f6597ed2fe8aa18092c369f88ff5dcfb2a",
 39    "25B01044_35_29_GT_cells.tif": "b690869bb27af6168cec0cd829db0e92926928bce5a5f9df9eb4be04cb01c9c2",
 40    "25B01044_35_29_GT_nuclei.tif": "02ad8aac22fae9b176109f7f61d3fa7ac4b153a2a84cacace545e34edaf5a379",
 41    "25B01044_35_29.tif": "c5656ce061df8bcd47558407664a776f357609319ca8d999f4f13e434caa17c6",
 42    "Ctrl_14_21_GT_cells.tif": "133f913f5839d9dfe060de78ea808e5896e5db0cdf33301add7610a9ec100914",
 43    "Ctrl_14_21_GT_nuclei.tif": "4be1554128c115d9023b3a4ac712ae88c68256f8a53c946363d5da523123fd36",
 44    "Ctrl_14_21.tif": "461b1028c7ad184d49b246101ed6b1348e8952825f3aaf2da52bab568482c4e9",
 45    "Ctrl_16_43_GT_cells.tif": "2edf66e06146a6397549f638161eebe669806a1b7ce44784bfee8537a388d2fe",
 46    "Ctrl_16_43_GT_nuclei.tif": "f88769584ac56e56ad6576ee96496ef2e726efce0af370a3ee905b4c8488f721",
 47    "Ctrl_16_43.tif": "ef59047b5efc430e96c1aa4061e2a18e50b80e30e2b221fd611ca248c234d229",
 48    "Panel33C_17_11_GT_cells.tif": "22267ab8423b180aff01ff7977439be1c96351e1755c59a7d10aefad48ac4aa4",
 49    "Panel33C_17_11_GT_nuclei.tif": "04738d4e3f07bf1e60ad70815ce6813033d2344fc728ae3256e00ed047692620",
 50    "Panel33C_17_11.tif": "e7fe7ecdac0adc658dcf8fbbc87db7921f72b43511c32e0af182088e0b0e1163",
 51    "Panel33C_22_26_GT_cells.tif": "f8558e71ed10020009c864bb4c4d60a4731b41ef9c429d1d7c700f59863395d2",
 52    "Panel33C_22_26_GT_nuclei.tif": "ee51216b7ccbca2ba7a4754c10401e01948a242553aac896e2c1a199341d6520",
 53    "Panel33C_22_26.tif": "1013423eafb9e8e73db31e9678a33a8f569dbdd519e30272ab5eaca6c6b8df37",
 54    "Panel33C_26_7_GT_cells.tif": "769b9143234dfa1bd9958ffd550f91f974e4c02c2f1da41e96a54b06140500d5",
 55    "Panel33C_26_7_GT_nuclei.tif": "a9c5b89e3abcffa1ef6d1ad680ccbee22c8a0a5f0818be0021a2f5025eb14fd8",
 56    "Panel33C_26_7.tif": "5909843c673515b4a500f72fca96a60dc12183d4f89ec963833fd79bf19f6356",
 57}
 58
 59LABEL_TYPES = ["nuclei", "cells"]
 60
 61
 62def get_mif_tonsil_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 63    """Download the Multiplex IF Tonsil dataset.
 64
 65    Args:
 66        path: Filepath to a folder where the downloaded data will be saved.
 67        download: Whether to download the data if it is not present.
 68
 69    Returns:
 70        The filepath to the downloaded data.
 71    """
 72    os.makedirs(path, exist_ok=True)
 73    for fname, checksum in CHECKSUMS.items():
 74        util.download_source(
 75            path=os.path.join(path, fname), url=f"{BASE_URL}/{fname}/content", download=download, checksum=checksum,
 76        )
 77
 78    return path
 79
 80
 81def get_mif_tonsil_paths(
 82    path: Union[os.PathLike, str], label_type: Literal["nuclei", "cells"] = "nuclei", download: bool = False,
 83) -> Tuple[List[str], List[str]]:
 84    """Get paths to the Multiplex IF Tonsil data.
 85
 86    Args:
 87        path: Filepath to a folder where the downloaded data will be saved.
 88        label_type: The choice of labels. Either 'nuclei' or 'cells'.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        List of filepaths for the image data.
 93        List of filepaths for the label data.
 94    """
 95    if label_type not in LABEL_TYPES:
 96        raise ValueError(f"'{label_type}' is not a valid label type. Choose one of {LABEL_TYPES}.")
 97
 98    data_dir = get_mif_tonsil_data(path, download)
 99
100    label_paths = natsorted(glob(os.path.join(data_dir, f"*_GT_{label_type}.tif")))
101    raw_paths = [p.replace(f"_GT_{label_type}.tif", ".tif") for p in label_paths]
102
103    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
104    assert all(os.path.exists(p) for p in raw_paths)
105
106    return raw_paths, label_paths
107
108
109def get_mif_tonsil_dataset(
110    path: Union[os.PathLike, str],
111    patch_shape: Tuple[int, int],
112    label_type: Literal["nuclei", "cells"] = "nuclei",
113    download: bool = False,
114    **kwargs
115) -> Dataset:
116    """Get the Multiplex IF Tonsil dataset for nucleus and cell instance segmentation.
117
118    Args:
119        path: Filepath to a folder where the downloaded data will be saved.
120        patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
121        label_type: The choice of labels. Either 'nuclei' or 'cells'.
122        download: Whether to download the data if it is not present.
123        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
124
125    Returns:
126        The segmentation dataset.
127    """
128    raw_paths, label_paths = get_mif_tonsil_paths(path, label_type, download)
129
130    return torch_em.default_segmentation_dataset(
131        raw_paths=raw_paths,
132        raw_key=None,
133        label_paths=label_paths,
134        label_key=None,
135        patch_shape=patch_shape,
136        with_channels=True,
137        is_seg_dataset=True,
138        ndim=2,
139        **kwargs
140    )
141
142
143def get_mif_tonsil_loader(
144    path: Union[os.PathLike, str],
145    batch_size: int,
146    patch_shape: Tuple[int, int],
147    label_type: Literal["nuclei", "cells"] = "nuclei",
148    download: bool = False,
149    **kwargs
150) -> DataLoader:
151    """Get the Multiplex IF Tonsil dataloader for nucleus and cell instance segmentation.
152
153    Args:
154        path: Filepath to a folder where the downloaded data will be saved.
155        batch_size: The batch size for training.
156        patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
157        label_type: The choice of labels. Either 'nuclei' or 'cells'.
158        download: Whether to download the data if it is not present.
159        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
160
161    Returns:
162        The DataLoader.
163    """
164    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
165    dataset = get_mif_tonsil_dataset(path, patch_shape, label_type, download, **ds_kwargs)
166    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://zenodo.org/api/records/22107836/files'
CHECKSUMS = {'23B10981_1_5_GT_cells.tif': 'e14a640cb0a9d0bfa05c6f4b8418a0dee498b372afdc0f557e37c9f22822a764', '23B10981_1_5_GT_nuclei.tif': '9473b9406afff196ac932da6692a73748fbb9c20f4681ac9708723ea6c725d4e', '23B10981_1_5.tif': '9f04251bd4d13210343fcc36b41d3ffb59b248897f80a2018e9568d81c4f050f', '23B10981_9_15_GT_cells.tif': '5a9c323c2a181e76318e9d2b07b91e815a7443b9b16318561e538d72858e5b95', '23B10981_9_15_GT_nuclei.tif': '06afc949650b5be542ded08a3e6daae5e3c39dc91db0c7d6f7059b6c00d53dfd', '23B10981_9_15.tif': 'f68d759e7ef734f30077333380132fb0bf82ed7c92efb96c581c9b2e383a1f0f', '24B2274_43_47_GT_cells.tif': '83e8b0810221d5fa3e208d2f8a27408c93bf52120a746d7a3840dd62945d2584', '24B2274_43_47_GT_nuclei.tif': '8c0871b27dc5bb2a3ab1443cb12011f7ee5f3f7dffb7e1d67b07499501a41d5f', '24B2274_43_47.tif': 'b9b3a0f37089819c7604294fe97904660ff5097454feb5d7bea5c3a1fde25abc', '24B2274_45_30_GT_cells.tif': '6de91c905770ac92c8b115a251657e42fe17ca647b7fb0572f5b3e12e8926cd4', '24B2274_45_30_GT_nuclei.tif': '4e43d9c4e27be42bac4b3ab8c29adff6d1d6879887fcccff5ce74e222b635ac2', '24B2274_45_30.tif': '9f98bdd7d158221ba84d5c5b3a3c22f6597ed2fe8aa18092c369f88ff5dcfb2a', '25B01044_35_29_GT_cells.tif': 'b690869bb27af6168cec0cd829db0e92926928bce5a5f9df9eb4be04cb01c9c2', '25B01044_35_29_GT_nuclei.tif': '02ad8aac22fae9b176109f7f61d3fa7ac4b153a2a84cacace545e34edaf5a379', '25B01044_35_29.tif': 'c5656ce061df8bcd47558407664a776f357609319ca8d999f4f13e434caa17c6', 'Ctrl_14_21_GT_cells.tif': '133f913f5839d9dfe060de78ea808e5896e5db0cdf33301add7610a9ec100914', 'Ctrl_14_21_GT_nuclei.tif': '4be1554128c115d9023b3a4ac712ae88c68256f8a53c946363d5da523123fd36', 'Ctrl_14_21.tif': '461b1028c7ad184d49b246101ed6b1348e8952825f3aaf2da52bab568482c4e9', 'Ctrl_16_43_GT_cells.tif': '2edf66e06146a6397549f638161eebe669806a1b7ce44784bfee8537a388d2fe', 'Ctrl_16_43_GT_nuclei.tif': 'f88769584ac56e56ad6576ee96496ef2e726efce0af370a3ee905b4c8488f721', 'Ctrl_16_43.tif': 'ef59047b5efc430e96c1aa4061e2a18e50b80e30e2b221fd611ca248c234d229', 'Panel33C_17_11_GT_cells.tif': '22267ab8423b180aff01ff7977439be1c96351e1755c59a7d10aefad48ac4aa4', 'Panel33C_17_11_GT_nuclei.tif': '04738d4e3f07bf1e60ad70815ce6813033d2344fc728ae3256e00ed047692620', 'Panel33C_17_11.tif': 'e7fe7ecdac0adc658dcf8fbbc87db7921f72b43511c32e0af182088e0b0e1163', 'Panel33C_22_26_GT_cells.tif': 'f8558e71ed10020009c864bb4c4d60a4731b41ef9c429d1d7c700f59863395d2', 'Panel33C_22_26_GT_nuclei.tif': 'ee51216b7ccbca2ba7a4754c10401e01948a242553aac896e2c1a199341d6520', 'Panel33C_22_26.tif': '1013423eafb9e8e73db31e9678a33a8f569dbdd519e30272ab5eaca6c6b8df37', 'Panel33C_26_7_GT_cells.tif': '769b9143234dfa1bd9958ffd550f91f974e4c02c2f1da41e96a54b06140500d5', 'Panel33C_26_7_GT_nuclei.tif': 'a9c5b89e3abcffa1ef6d1ad680ccbee22c8a0a5f0818be0021a2f5025eb14fd8', 'Panel33C_26_7.tif': '5909843c673515b4a500f72fca96a60dc12183d4f89ec963833fd79bf19f6356'}
LABEL_TYPES = ['nuclei', 'cells']
def get_mif_tonsil_data(path: Union[os.PathLike, str], download: bool = False) -> str:
63def get_mif_tonsil_data(path: Union[os.PathLike, str], download: bool = False) -> str:
64    """Download the Multiplex IF Tonsil dataset.
65
66    Args:
67        path: Filepath to a folder where the downloaded data will be saved.
68        download: Whether to download the data if it is not present.
69
70    Returns:
71        The filepath to the downloaded data.
72    """
73    os.makedirs(path, exist_ok=True)
74    for fname, checksum in CHECKSUMS.items():
75        util.download_source(
76            path=os.path.join(path, fname), url=f"{BASE_URL}/{fname}/content", download=download, checksum=checksum,
77        )
78
79    return path

Download the Multiplex IF Tonsil dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • download: Whether to download the data if it is not present.
Returns:

The filepath to the downloaded data.

def get_mif_tonsil_paths( path: Union[os.PathLike, str], label_type: Literal['nuclei', 'cells'] = 'nuclei', download: bool = False) -> Tuple[List[str], List[str]]:
 82def get_mif_tonsil_paths(
 83    path: Union[os.PathLike, str], label_type: Literal["nuclei", "cells"] = "nuclei", download: bool = False,
 84) -> Tuple[List[str], List[str]]:
 85    """Get paths to the Multiplex IF Tonsil data.
 86
 87    Args:
 88        path: Filepath to a folder where the downloaded data will be saved.
 89        label_type: The choice of labels. Either 'nuclei' or 'cells'.
 90        download: Whether to download the data if it is not present.
 91
 92    Returns:
 93        List of filepaths for the image data.
 94        List of filepaths for the label data.
 95    """
 96    if label_type not in LABEL_TYPES:
 97        raise ValueError(f"'{label_type}' is not a valid label type. Choose one of {LABEL_TYPES}.")
 98
 99    data_dir = get_mif_tonsil_data(path, download)
100
101    label_paths = natsorted(glob(os.path.join(data_dir, f"*_GT_{label_type}.tif")))
102    raw_paths = [p.replace(f"_GT_{label_type}.tif", ".tif") for p in label_paths]
103
104    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
105    assert all(os.path.exists(p) for p in raw_paths)
106
107    return raw_paths, label_paths

Get paths to the Multiplex IF Tonsil data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • label_type: The choice of labels. Either 'nuclei' or 'cells'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_mif_tonsil_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], label_type: Literal['nuclei', 'cells'] = 'nuclei', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
110def get_mif_tonsil_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    label_type: Literal["nuclei", "cells"] = "nuclei",
114    download: bool = False,
115    **kwargs
116) -> Dataset:
117    """Get the Multiplex IF Tonsil dataset for nucleus and cell instance segmentation.
118
119    Args:
120        path: Filepath to a folder where the downloaded data will be saved.
121        patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
122        label_type: The choice of labels. Either 'nuclei' or 'cells'.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    raw_paths, label_paths = get_mif_tonsil_paths(path, label_type, download)
130
131    return torch_em.default_segmentation_dataset(
132        raw_paths=raw_paths,
133        raw_key=None,
134        label_paths=label_paths,
135        label_key=None,
136        patch_shape=patch_shape,
137        with_channels=True,
138        is_seg_dataset=True,
139        ndim=2,
140        **kwargs
141    )

Get the Multiplex IF Tonsil dataset for nucleus and cell instance segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
  • label_type: The choice of labels. Either 'nuclei' or 'cells'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_mif_tonsil_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], label_type: Literal['nuclei', 'cells'] = 'nuclei', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
144def get_mif_tonsil_loader(
145    path: Union[os.PathLike, str],
146    batch_size: int,
147    patch_shape: Tuple[int, int],
148    label_type: Literal["nuclei", "cells"] = "nuclei",
149    download: bool = False,
150    **kwargs
151) -> DataLoader:
152    """Get the Multiplex IF Tonsil dataloader for nucleus and cell instance segmentation.
153
154    Args:
155        path: Filepath to a folder where the downloaded data will be saved.
156        batch_size: The batch size for training.
157        patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
158        label_type: The choice of labels. Either 'nuclei' or 'cells'.
159        download: Whether to download the data if it is not present.
160        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
161
162    Returns:
163        The DataLoader.
164    """
165    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
166    dataset = get_mif_tonsil_dataset(path, patch_shape, label_type, download, **ds_kwargs)
167    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Multiplex IF Tonsil dataloader for nucleus and cell instance segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training. The images have a size of 100x100 pixels.
  • label_type: The choice of labels. Either 'nuclei' or 'cells'.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.