torch_em.data.datasets.light_microscopy.multiphoton_liver

The Multiphoton Liver dataset contains annotations for 3D segmentation of cell borders, bile canaliculi (BC), sinusoids and nuclei in multiphoton microscopy volumes of liver tissue.

The dataset is a resource for benchmarking 3D microscopy image-analysis methods, published on the BioImage Archive as accession S-BIAD2705. It contains four real microscopy samples ('G1', 'G2', 'Ch2', 'Ch3'), each a five-channel volume (raw fluorescence channels, meaning undocumented) paired with a four-channel segmentation mask. Per the dataset's own 'Naming_Convention.md', the mask channels are, in order: BC, Sinusoids, Nuclei, Sinusoid Fill. Nuclei are already instance-labeled by the authors (verified: up to several thousand distinct ids per volume); the other three channels are binary semantic masks (verified: only two distinct values per channel). Use target to select which of the four to expose as the label ('bc', 'sinusoids', 'nuclei' or 'sinusoid_fill').

NOTE: The full BioImage Archive resource is about 500 GB across roughly 23,000 files, including simulated data, restoration and segmentation baselines. This module only downloads the four raw 'Preprocessed' microscopy volumes and their four 'Microscopy_segmented_mask' files (about 7-8 GB per sample). Use samples to further restrict to a subset. Only the checksum of the 'G2' sample's two files is known (computed from a verified download); the 'G1', 'Ch2' and 'Ch3' files have no published checksum and are downloaded without one.

The data is located at https://www.ebi.ac.uk/biostudies/BioImages/studies/S-BIAD2705, released under a CC-BY-4.0 license (per the BioStudies study page).

  1"""The Multiphoton Liver dataset contains annotations for 3D segmentation of cell borders, bile canaliculi (BC),
  2sinusoids and nuclei in multiphoton microscopy volumes of liver tissue.
  3
  4The dataset is a resource for benchmarking 3D microscopy image-analysis methods, published on the BioImage Archive
  5as accession S-BIAD2705. It contains four real microscopy samples ('G1', 'G2', 'Ch2', 'Ch3'), each a five-channel
  6volume (raw fluorescence channels, meaning undocumented) paired with a four-channel segmentation mask. Per the
  7dataset's own 'Naming_Convention.md', the mask channels are, in order: BC, Sinusoids, Nuclei, Sinusoid Fill.
  8Nuclei are already instance-labeled by the authors (verified: up to several thousand distinct ids per volume);
  9the other three channels are binary semantic masks (verified: only two distinct values per channel). Use `target`
 10to select which of the four to expose as the label ('bc', 'sinusoids', 'nuclei' or 'sinusoid_fill').
 11
 12NOTE: The full BioImage Archive resource is about 500 GB across roughly 23,000 files, including simulated data,
 13restoration and segmentation baselines. This module only downloads the four raw 'Preprocessed' microscopy volumes
 14and their four 'Microscopy_segmented_mask' files (about 7-8 GB per sample). Use `samples` to further restrict to
 15a subset. Only the checksum of the 'G2' sample's two files is known (computed from a verified download); the
 16'G1', 'Ch2' and 'Ch3' files have no published checksum and are downloaded without one.
 17
 18The data is located at https://www.ebi.ac.uk/biostudies/BioImages/studies/S-BIAD2705, released under a CC-BY-4.0
 19license (per the BioStudies study page).
 20"""
 21
 22import os
 23from glob import glob
 24from natsort import natsorted
 25from typing import Union, Tuple, Optional, Sequence, List, Literal
 26
 27from torch.utils.data import Dataset, DataLoader
 28
 29import torch_em
 30
 31from .. import util
 32
 33
 34BASE_URL = "https://ftp.ebi.ac.uk/biostudies/fire/S-BIAD/705/S-BIAD2705/Files"
 35
 36SAMPLES = {
 37    "G1": ("Dataset/Image_sets/Microscopy_images/Preprocessed/20221014_6_Control(G1).tif",
 38           "Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_G1_segmented_mask.tif", None, None),
 39    "G2": ("Dataset/Image_sets/Microscopy_images/Preprocessed/20221025_29_Control (G2).tif",
 40           "Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_G2_segmented_mask.tif",
 41           "531e2f00db607760bc7587c6391db3833ea2de9cbf930597c9ef2b97c942bcd5",
 42           "60c1a9fd43a9f99550ce994dc825cb3e498c514a667116977b1a9c04d70f6a93"),
 43    "Ch2": ("Dataset/Image_sets/Microscopy_images/Preprocessed/20240913_control_c1 (Ch2).tif",
 44            "Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_Ch2_segmented_mask.tif", None, None),
 45    "Ch3": ("Dataset/Image_sets/Microscopy_images/Preprocessed/20241029_control (ch3).tif",
 46            "Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_Ch3_segmented_mask.tif", None, None),
 47}
 48"""Mapping from sample name to (image path, mask path, image sha256, mask sha256) relative to `BASE_URL`."""
 49
 50MASK_CHANNELS = {"bc": 0, "sinusoids": 1, "nuclei": 2, "sinusoid_fill": 3}
 51
 52
 53def _preprocess_sample(name, path, preprocessed_dir):
 54    import h5py
 55    import tifffile
 56
 57    h5_path = os.path.join(preprocessed_dir, f"{name}.h5")
 58    if os.path.exists(h5_path):
 59        return
 60
 61    image_rel, mask_rel, _, _ = SAMPLES[name]
 62    raw = tifffile.imread(os.path.join(path, f"{name}_image.tif"))
 63    mask = tifffile.imread(os.path.join(path, f"{name}_mask.tif"))
 64    assert raw.shape[1:] == mask.shape[1:], f"Shape mismatch for {name}: raw={raw.shape}, mask={mask.shape}"
 65
 66    tmp_path = f"{h5_path}.{os.getpid()}.incomplete"
 67    with h5py.File(tmp_path, "w") as f:
 68        f.create_dataset("raw", data=raw, compression="gzip", chunks=(1,) + raw.shape[1:])
 69        for target, channel in MASK_CHANNELS.items():
 70            data = mask[channel] if target == "nuclei" else (mask[channel] > 0).astype("uint8")
 71            f.create_dataset(f"labels/{target}", data=data, compression="gzip", chunks=raw.shape[1:])
 72    os.replace(tmp_path, h5_path)
 73
 74
 75def get_multiphoton_liver_data(
 76    path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False,
 77) -> str:
 78    """Download the Multiphoton Liver dataset and convert it to hdf5 files.
 79
 80    Args:
 81        path: Filepath to a folder where the downloaded data will be saved.
 82        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
 83        download: Whether to download the data if it is not present.
 84
 85    Returns:
 86        Filepath to the folder with the preprocessed data.
 87    """
 88    samples = list(SAMPLES) if samples is None else samples
 89    invalid = set(samples) - set(SAMPLES)
 90    if invalid:
 91        raise ValueError(f"{invalid} are not valid samples. Choose from {list(SAMPLES)}.")
 92
 93    preprocessed_dir = os.path.join(path, "preprocessed")
 94    os.makedirs(preprocessed_dir, exist_ok=True)
 95
 96    for name in samples:
 97        if os.path.exists(os.path.join(preprocessed_dir, f"{name}.h5")):
 98            continue
 99
100        image_rel, mask_rel, image_checksum, mask_checksum = SAMPLES[name]
101        image_path = os.path.join(path, f"{name}_image.tif")
102        mask_path = os.path.join(path, f"{name}_mask.tif")
103        util.download_source(
104            path=image_path, url=f"{BASE_URL}/{image_rel}", download=download, checksum=image_checksum
105        )
106        util.download_source(
107            path=mask_path, url=f"{BASE_URL}/{mask_rel}", download=download, checksum=mask_checksum
108        )
109        _preprocess_sample(name, path, preprocessed_dir)
110
111    return preprocessed_dir
112
113
114def get_multiphoton_liver_paths(
115    path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False,
116) -> List[str]:
117    """Get paths to the Multiphoton Liver data.
118
119    Args:
120        path: Filepath to a folder where the downloaded data will be saved.
121        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
122        download: Whether to download the data if it is not present.
123
124    Returns:
125        List of filepaths for the hdf5 files, which contain the raw data ('raw') and the label data
126        ('labels/bc', 'labels/sinusoids', 'labels/nuclei', 'labels/sinusoid_fill').
127    """
128    data_dir = get_multiphoton_liver_data(path, samples, download)
129    paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
130    if samples is not None:
131        paths = [p for p in paths if os.path.splitext(os.path.basename(p))[0] in samples]
132    assert len(paths) > 0
133    return paths
134
135
136def get_multiphoton_liver_dataset(
137    path: Union[os.PathLike, str],
138    patch_shape: Tuple[int, int, int],
139    target: Literal["bc", "sinusoids", "nuclei", "sinusoid_fill"] = "nuclei",
140    samples: Optional[Sequence[str]] = None,
141    download: bool = False,
142    **kwargs
143) -> Dataset:
144    """Get the Multiphoton Liver dataset for 3D segmentation of liver tissue structures.
145
146    Args:
147        path: Filepath to a folder where the downloaded data will be saved.
148        patch_shape: The patch shape to use for training.
149        target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'.
150            Only 'nuclei' is an instance segmentation target, the others are binary.
151        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
152        download: Whether to download the data if it is not present.
153        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
154
155    Returns:
156        The segmentation dataset.
157    """
158    if target not in MASK_CHANNELS:
159        raise ValueError(f"'{target}' is not a valid target. Choose one of {list(MASK_CHANNELS)}.")
160
161    data_paths = get_multiphoton_liver_paths(path, samples, download)
162
163    return torch_em.default_segmentation_dataset(
164        raw_paths=data_paths,
165        raw_key="raw",
166        label_paths=data_paths,
167        label_key=f"labels/{target}",
168        patch_shape=patch_shape,
169        with_channels=True,
170        is_seg_dataset=True,
171        ndim=3,
172        **kwargs
173    )
174
175
176def get_multiphoton_liver_loader(
177    path: Union[os.PathLike, str],
178    batch_size: int,
179    patch_shape: Tuple[int, int, int],
180    target: Literal["bc", "sinusoids", "nuclei", "sinusoid_fill"] = "nuclei",
181    samples: Optional[Sequence[str]] = None,
182    download: bool = False,
183    **kwargs
184) -> DataLoader:
185    """Get the Multiphoton Liver dataloader for 3D segmentation of liver tissue structures.
186
187    Args:
188        path: Filepath to a folder where the downloaded data will be saved.
189        batch_size: The batch size for training.
190        patch_shape: The patch shape to use for training.
191        target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'.
192        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
193        download: Whether to download the data if it is not present.
194        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
195
196    Returns:
197        The DataLoader.
198    """
199    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
200    dataset = get_multiphoton_liver_dataset(path, patch_shape, target, samples, download, **ds_kwargs)
201    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://ftp.ebi.ac.uk/biostudies/fire/S-BIAD/705/S-BIAD2705/Files'
SAMPLES = {'G1': ('Dataset/Image_sets/Microscopy_images/Preprocessed/20221014_6_Control(G1).tif', 'Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_G1_segmented_mask.tif', None, None), 'G2': ('Dataset/Image_sets/Microscopy_images/Preprocessed/20221025_29_Control (G2).tif', 'Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_G2_segmented_mask.tif', '531e2f00db607760bc7587c6391db3833ea2de9cbf930597c9ef2b97c942bcd5', '60c1a9fd43a9f99550ce994dc825cb3e498c514a667116977b1a9c04d70f6a93'), 'Ch2': ('Dataset/Image_sets/Microscopy_images/Preprocessed/20240913_control_c1 (Ch2).tif', 'Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_Ch2_segmented_mask.tif', None, None), 'Ch3': ('Dataset/Image_sets/Microscopy_images/Preprocessed/20241029_control (ch3).tif', 'Dataset/Mask_sets/Microscopy_segmented_masks/Microscopy_Ch3_segmented_mask.tif', None, None)}

Mapping from sample name to (image path, mask path, image sha256, mask sha256) relative to BASE_URL.

MASK_CHANNELS = {'bc': 0, 'sinusoids': 1, 'nuclei': 2, 'sinusoid_fill': 3}
def get_multiphoton_liver_data( path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False) -> str:
 76def get_multiphoton_liver_data(
 77    path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False,
 78) -> str:
 79    """Download the Multiphoton Liver dataset and convert it to hdf5 files.
 80
 81    Args:
 82        path: Filepath to a folder where the downloaded data will be saved.
 83        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
 84        download: Whether to download the data if it is not present.
 85
 86    Returns:
 87        Filepath to the folder with the preprocessed data.
 88    """
 89    samples = list(SAMPLES) if samples is None else samples
 90    invalid = set(samples) - set(SAMPLES)
 91    if invalid:
 92        raise ValueError(f"{invalid} are not valid samples. Choose from {list(SAMPLES)}.")
 93
 94    preprocessed_dir = os.path.join(path, "preprocessed")
 95    os.makedirs(preprocessed_dir, exist_ok=True)
 96
 97    for name in samples:
 98        if os.path.exists(os.path.join(preprocessed_dir, f"{name}.h5")):
 99            continue
100
101        image_rel, mask_rel, image_checksum, mask_checksum = SAMPLES[name]
102        image_path = os.path.join(path, f"{name}_image.tif")
103        mask_path = os.path.join(path, f"{name}_mask.tif")
104        util.download_source(
105            path=image_path, url=f"{BASE_URL}/{image_rel}", download=download, checksum=image_checksum
106        )
107        util.download_source(
108            path=mask_path, url=f"{BASE_URL}/{mask_rel}", download=download, checksum=mask_checksum
109        )
110        _preprocess_sample(name, path, preprocessed_dir)
111
112    return preprocessed_dir

Download the Multiphoton Liver dataset and convert it to hdf5 files.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • samples: The sample names to use, see SAMPLES. By default all four samples are used.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the preprocessed data.

def get_multiphoton_liver_paths( path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False) -> List[str]:
115def get_multiphoton_liver_paths(
116    path: Union[os.PathLike, str], samples: Optional[Sequence[str]] = None, download: bool = False,
117) -> List[str]:
118    """Get paths to the Multiphoton Liver data.
119
120    Args:
121        path: Filepath to a folder where the downloaded data will be saved.
122        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
123        download: Whether to download the data if it is not present.
124
125    Returns:
126        List of filepaths for the hdf5 files, which contain the raw data ('raw') and the label data
127        ('labels/bc', 'labels/sinusoids', 'labels/nuclei', 'labels/sinusoid_fill').
128    """
129    data_dir = get_multiphoton_liver_data(path, samples, download)
130    paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
131    if samples is not None:
132        paths = [p for p in paths if os.path.splitext(os.path.basename(p))[0] in samples]
133    assert len(paths) > 0
134    return paths

Get paths to the Multiphoton Liver data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • samples: The sample names to use, see SAMPLES. By default all four samples are used.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the raw data ('raw') and the label data ('labels/bc', 'labels/sinusoids', 'labels/nuclei', 'labels/sinusoid_fill').

def get_multiphoton_liver_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], target: Literal['bc', 'sinusoids', 'nuclei', 'sinusoid_fill'] = 'nuclei', samples: Optional[Sequence[str]] = None, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
137def get_multiphoton_liver_dataset(
138    path: Union[os.PathLike, str],
139    patch_shape: Tuple[int, int, int],
140    target: Literal["bc", "sinusoids", "nuclei", "sinusoid_fill"] = "nuclei",
141    samples: Optional[Sequence[str]] = None,
142    download: bool = False,
143    **kwargs
144) -> Dataset:
145    """Get the Multiphoton Liver dataset for 3D segmentation of liver tissue structures.
146
147    Args:
148        path: Filepath to a folder where the downloaded data will be saved.
149        patch_shape: The patch shape to use for training.
150        target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'.
151            Only 'nuclei' is an instance segmentation target, the others are binary.
152        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
155
156    Returns:
157        The segmentation dataset.
158    """
159    if target not in MASK_CHANNELS:
160        raise ValueError(f"'{target}' is not a valid target. Choose one of {list(MASK_CHANNELS)}.")
161
162    data_paths = get_multiphoton_liver_paths(path, samples, download)
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=data_paths,
166        raw_key="raw",
167        label_paths=data_paths,
168        label_key=f"labels/{target}",
169        patch_shape=patch_shape,
170        with_channels=True,
171        is_seg_dataset=True,
172        ndim=3,
173        **kwargs
174    )

Get the Multiphoton Liver dataset for 3D segmentation of liver tissue structures.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training.
  • target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'. Only 'nuclei' is an instance segmentation target, the others are binary.
  • samples: The sample names to use, see SAMPLES. By default all four samples are used.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_multiphoton_liver_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], target: Literal['bc', 'sinusoids', 'nuclei', 'sinusoid_fill'] = 'nuclei', samples: Optional[Sequence[str]] = None, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
177def get_multiphoton_liver_loader(
178    path: Union[os.PathLike, str],
179    batch_size: int,
180    patch_shape: Tuple[int, int, int],
181    target: Literal["bc", "sinusoids", "nuclei", "sinusoid_fill"] = "nuclei",
182    samples: Optional[Sequence[str]] = None,
183    download: bool = False,
184    **kwargs
185) -> DataLoader:
186    """Get the Multiphoton Liver dataloader for 3D segmentation of liver tissue structures.
187
188    Args:
189        path: Filepath to a folder where the downloaded data will be saved.
190        batch_size: The batch size for training.
191        patch_shape: The patch shape to use for training.
192        target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'.
193        samples: The sample names to use, see `SAMPLES`. By default all four samples are used.
194        download: Whether to download the data if it is not present.
195        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
196
197    Returns:
198        The DataLoader.
199    """
200    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
201    dataset = get_multiphoton_liver_dataset(path, patch_shape, target, samples, download, **ds_kwargs)
202    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Multiphoton Liver dataloader for 3D segmentation of liver tissue structures.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • target: The choice of segmentation target. One of 'bc', 'sinusoids', 'nuclei' or 'sinusoid_fill'.
  • samples: The sample names to use, see SAMPLES. By default all four samples are used.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.