torch_em.data.datasets.electron_microscopy.fluoem
The FluoEM dataset contains an electron microscopy volume of mouse cortex with proofread axon instance segmentation, generated to validate a fluorescence-guided axon labeling method (FluoEM) for long-range connectomics.
Segmentation is dense but not uniformly proofread throughout the volume: some regions still show block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches at internal block boundaries.
The data is available via the Helmstaedter Lab's public WEBKNOSSOS instance (organization MPI_Brain_Research). No license is specified by the data provider. The dataset was published in https://doi.org/10.7554/eLife.38976. Please cite this publication if you use the dataset in your research.
1"""The FluoEM dataset contains an electron microscopy volume of mouse cortex with proofread 2axon instance segmentation, generated to validate a fluorescence-guided axon labeling method 3(FluoEM) for long-range connectomics. 4 5Segmentation is dense but not uniformly proofread throughout the volume: some regions still show 6block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches 7at internal block boundaries. 8 9The data is available via the Helmstaedter Lab's public WEBKNOSSOS instance (organization 10MPI_Brain_Research). No license is specified by the data provider. 11The dataset was published in https://doi.org/10.7554/eLife.38976. 12Please cite this publication if you use the dataset in your research. 13""" 14 15import os 16from typing import Tuple, List, Union 17 18import numpy as np 19 20from torch.utils.data import DataLoader, Dataset 21 22import torch_em 23 24from .. import util 25 26 27DATASTORE_URL = "https://demo.wk1.connectomics.hpccloud.mpg.de/data/zarr" 28DATASET_ID = "5abb89f348d7a73cea448019" 29SHAPE_ZYX = (2767, 26242, 29000) 30RESOLUTION_NM = (30, 11.24, 11.24) 31 32 33def _bbox_to_str(bounding_box): 34 import hashlib 35 return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12] 36 37 38def _open_remote_array(layer): 39 import fsspec 40 import zarr 41 return zarr.open(fsspec.get_mapper(f"{DATASTORE_URL}/{DATASET_ID}/{layer}/1-1-1"), mode="r", zarr_format=2) 42 43 44def get_fluoem_data( 45 path: Union[os.PathLike, str], 46 bounding_box: Tuple[int, int, int, int, int, int], 47 download: bool = False, 48) -> str: 49 """Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store. 50 51 Args: 52 path: Filepath to a folder where the cached zarr store will be saved. 53 bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel 54 coordinates, at the dataset's native resolution. 55 download: Whether to stream and cache the data if it is not present. 56 57 Returns: 58 The filepath to the cached zarr store. 59 """ 60 import zarr 61 from zarr.codecs import BloscCodec 62 63 os.makedirs(str(path), exist_ok=True) 64 zarr_path = os.path.join(str(path), f"{_bbox_to_str(bounding_box)}.zarr") 65 66 root = zarr.open_group(zarr_path, mode="a") 67 if "raw" in root and "labels" in root: 68 return zarr_path 69 70 if not download: 71 raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.") 72 73 z_min, z_max, y_min, y_max, x_min, x_max = bounding_box 74 shape_z, shape_y, shape_x = SHAPE_ZYX 75 assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \ 76 f"Bounding box exceeds the FluoEM volume {SHAPE_ZYX}" 77 78 color = _open_remote_array("color") 79 seg = _open_remote_array("segmentation") 80 81 # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x). 82 raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 83 labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 84 85 def _make_array(name, data, shuffle): 86 array = root.create_array( 87 name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype, 88 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 89 ) 90 array[:] = data 91 92 root.attrs["bounding_box"] = list(bounding_box) 93 root.attrs["resolution_nm"] = list(RESOLUTION_NM) 94 root.attrs["labels_are_exhaustive"] = False 95 96 _make_array("raw", raw, shuffle="shuffle") 97 _make_array("labels", labels, shuffle="bitshuffle") 98 99 return zarr_path 100 101 102def get_fluoem_paths( 103 path: Union[os.PathLike, str], 104 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 105 download: bool = False, 106) -> List[str]: 107 """Get paths to cached FluoEM zarr stores. 108 109 Args: 110 path: Filepath to a folder where the cached zarr stores will be saved. 111 bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max). 112 download: Whether to stream and cache the data if it is not present. 113 114 Returns: 115 List of filepaths to the cached zarr stores. 116 """ 117 return [get_fluoem_data(path, bbox, download) for bbox in bounding_boxes] 118 119 120def get_fluoem_dataset( 121 path: Union[os.PathLike, str], 122 patch_shape: Tuple[int, int, int], 123 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 124 download: bool = False, 125 **kwargs, 126) -> Dataset: 127 """Get the FluoEM dataset for axon instance segmentation. 128 129 Args: 130 path: Filepath to a folder where the cached zarr stores will be saved. 131 patch_shape: The patch shape (z, y, x) to use for training. 132 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 133 download: Whether to stream and cache data if not already present. 134 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 135 136 Returns: 137 The segmentation dataset. 138 """ 139 assert len(patch_shape) == 3 140 141 paths = get_fluoem_paths(path, bounding_boxes, download) 142 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 143 144 return torch_em.default_segmentation_dataset( 145 raw_paths=paths, 146 raw_key="raw", 147 label_paths=paths, 148 label_key="labels", 149 patch_shape=patch_shape, 150 **kwargs, 151 ) 152 153 154def get_fluoem_loader( 155 path: Union[os.PathLike, str], 156 patch_shape: Tuple[int, int, int], 157 batch_size: int, 158 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 159 download: bool = False, 160 **kwargs, 161) -> DataLoader: 162 """Get the DataLoader for axon instance segmentation in the FluoEM dataset. 163 164 Args: 165 path: Filepath to a folder where the cached zarr stores will be saved. 166 patch_shape: The patch shape (z, y, x) to use for training. 167 batch_size: The batch size for training. 168 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 169 download: Whether to stream and cache data if not already present. 170 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 171 PyTorch DataLoader. 172 173 Returns: 174 The DataLoader. 175 """ 176 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 177 dataset = get_fluoem_dataset(path, patch_shape, bounding_boxes, download, **ds_kwargs) 178 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
45def get_fluoem_data( 46 path: Union[os.PathLike, str], 47 bounding_box: Tuple[int, int, int, int, int, int], 48 download: bool = False, 49) -> str: 50 """Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store. 51 52 Args: 53 path: Filepath to a folder where the cached zarr store will be saved. 54 bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel 55 coordinates, at the dataset's native resolution. 56 download: Whether to stream and cache the data if it is not present. 57 58 Returns: 59 The filepath to the cached zarr store. 60 """ 61 import zarr 62 from zarr.codecs import BloscCodec 63 64 os.makedirs(str(path), exist_ok=True) 65 zarr_path = os.path.join(str(path), f"{_bbox_to_str(bounding_box)}.zarr") 66 67 root = zarr.open_group(zarr_path, mode="a") 68 if "raw" in root and "labels" in root: 69 return zarr_path 70 71 if not download: 72 raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.") 73 74 z_min, z_max, y_min, y_max, x_min, x_max = bounding_box 75 shape_z, shape_y, shape_x = SHAPE_ZYX 76 assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \ 77 f"Bounding box exceeds the FluoEM volume {SHAPE_ZYX}" 78 79 color = _open_remote_array("color") 80 seg = _open_remote_array("segmentation") 81 82 # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x). 83 raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 84 labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 85 86 def _make_array(name, data, shuffle): 87 array = root.create_array( 88 name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype, 89 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 90 ) 91 array[:] = data 92 93 root.attrs["bounding_box"] = list(bounding_box) 94 root.attrs["resolution_nm"] = list(RESOLUTION_NM) 95 root.attrs["labels_are_exhaustive"] = False 96 97 _make_array("raw", raw, shuffle="shuffle") 98 _make_array("labels", labels, shuffle="bitshuffle") 99 100 return zarr_path
Stream a subvolume of the FluoEM dataset and cache it as a zarr v3 store.
Arguments:
- path: Filepath to a folder where the cached zarr store will be saved.
- bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at the dataset's native resolution.
- download: Whether to stream and cache the data if it is not present.
Returns:
The filepath to the cached zarr store.
103def get_fluoem_paths( 104 path: Union[os.PathLike, str], 105 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 106 download: bool = False, 107) -> List[str]: 108 """Get paths to cached FluoEM zarr stores. 109 110 Args: 111 path: Filepath to a folder where the cached zarr stores will be saved. 112 bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max). 113 download: Whether to stream and cache the data if it is not present. 114 115 Returns: 116 List of filepaths to the cached zarr stores. 117 """ 118 return [get_fluoem_data(path, bbox, download) for bbox in bounding_boxes]
Get paths to cached FluoEM zarr stores.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- download: Whether to stream and cache the data if it is not present.
Returns:
List of filepaths to the cached zarr stores.
121def get_fluoem_dataset( 122 path: Union[os.PathLike, str], 123 patch_shape: Tuple[int, int, int], 124 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 125 download: bool = False, 126 **kwargs, 127) -> Dataset: 128 """Get the FluoEM dataset for axon instance segmentation. 129 130 Args: 131 path: Filepath to a folder where the cached zarr stores will be saved. 132 patch_shape: The patch shape (z, y, x) to use for training. 133 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 134 download: Whether to stream and cache data if not already present. 135 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 136 137 Returns: 138 The segmentation dataset. 139 """ 140 assert len(patch_shape) == 3 141 142 paths = get_fluoem_paths(path, bounding_boxes, download) 143 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 144 145 return torch_em.default_segmentation_dataset( 146 raw_paths=paths, 147 raw_key="raw", 148 label_paths=paths, 149 label_key="labels", 150 patch_shape=patch_shape, 151 **kwargs, 152 )
Get the FluoEM dataset for axon instance segmentation.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- download: Whether to stream and cache data if not already present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
155def get_fluoem_loader( 156 path: Union[os.PathLike, str], 157 patch_shape: Tuple[int, int, int], 158 batch_size: int, 159 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 160 download: bool = False, 161 **kwargs, 162) -> DataLoader: 163 """Get the DataLoader for axon instance segmentation in the FluoEM dataset. 164 165 Args: 166 path: Filepath to a folder where the cached zarr stores will be saved. 167 patch_shape: The patch shape (z, y, x) to use for training. 168 batch_size: The batch size for training. 169 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 170 download: Whether to stream and cache data if not already present. 171 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 172 PyTorch DataLoader. 173 174 Returns: 175 The DataLoader. 176 """ 177 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 178 dataset = get_fluoem_dataset(path, patch_shape, bounding_boxes, download, **ds_kwargs) 179 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the DataLoader for axon instance segmentation in the FluoEM dataset.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- batch_size: The batch size for training.
- bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- download: Whether to stream and cache data if not already present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.