torch_em.data.datasets.electron_microscopy.neuromast_connectomics
The NeuromastConnectomics dataset contains dense electron microscopy connectomic reconstructions of zebrafish lateral-line neuromasts, with proofread cell instance segmentation of hair cells and afferent/efferent neurons.
Ten of the paper's fourteen named neuromast volumes are included here; the remaining four were not found as separate public entries at the time of writing this module.
The data is available via the Hudspeth Lab's public WEBKNOSSOS instance (organization 'The Rockefeller University'). No license is specified by the data provider. The dataset was published in https://doi.org/10.7554/eLife.33988. Please cite this publication if you use the dataset in your research.
1"""The NeuromastConnectomics dataset contains dense electron microscopy connectomic 2reconstructions of zebrafish lateral-line neuromasts, with proofread cell instance 3segmentation of hair cells and afferent/efferent neurons. 4 5Ten of the paper's fourteen named neuromast volumes are included here; the remaining four 6were not found as separate public entries at the time of writing this module. 7 8The data is available via the Hudspeth Lab's public WEBKNOSSOS instance (organization 9'The Rockefeller University'). No license is specified by the data provider. 10The dataset was published in https://doi.org/10.7554/eLife.33988. 11Please cite this publication if you use the dataset in your research. 12""" 13 14import os 15from typing import Dict, List, Literal, Tuple, Union 16 17import numpy as np 18 19from torch.utils.data import DataLoader, Dataset 20 21import torch_em 22 23from .. import util 24 25 26DATASTORE_URL = "https://data-humerus.webknossos.org/data/zarr" 27 28SAMPLES: Dict[str, dict] = { 29 "wt1": {"dataset_id": "5b3ce7bb1d0000340240b921", "shape_zyx": (1024, 9216, 9216), "resolution_nm": (30, 6, 6)}, 30 "wt2": {"dataset_id": "5b3ce7bb1d0000340240b923", "shape_zyx": (1024, 10240, 13312), "resolution_nm": (50, 5, 5)}, 31 "wt6": {"dataset_id": "5b3ce7bb1d0000340240b91d", "shape_zyx": (2048, 10240, 11264), "resolution_nm": (30, 5, 5)}, 32 "wt7": {"dataset_id": "5b3ce7bb1d0000340240b91c", "shape_zyx": (1024, 8192, 10240), "resolution_nm": (30, 6, 6)}, 33 "wt8": {"dataset_id": "5b3ce7bb1d0000340240b91f", "shape_zyx": (2048, 11264, 11264), "resolution_nm": (30, 6, 6)}, 34 "t1": {"dataset_id": "5b3ce7bb1d0000340240b91b", "shape_zyx": (2048, 6144, 6144), "resolution_nm": (30, 6, 6)}, 35 "t3": {"dataset_id": "5b4238a61d0000c42a1ab5e3", "shape_zyx": (2048, 7168, 8192), "resolution_nm": (30, 6, 6)}, 36 "t4": {"dataset_id": "5b4238a61d0000c42a1ab5e5", "shape_zyx": (2048, 10240, 9216), "resolution_nm": (30, 6, 6)}, 37 "n1": {"dataset_id": "5b4238a61d00004b1d1ab5e2", "shape_zyx": (1024, 9216, 11264), "resolution_nm": (30, 6, 6)}, 38 "n2": {"dataset_id": "5b4238a61d0000c42a1ab5e4", "shape_zyx": (1024, 9216, 9216), "resolution_nm": (30, 6, 6)}, 39} 40 41 42def _bbox_to_str(bounding_box): 43 import hashlib 44 return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12] 45 46 47def _open_remote_array(dataset_id, layer): 48 import fsspec 49 import zarr 50 return zarr.open(fsspec.get_mapper(f"{DATASTORE_URL}/{dataset_id}/{layer}/1-1-1"), mode="r", zarr_format=2) 51 52 53def get_neuromast_connectomics_data( 54 path: Union[os.PathLike, str], 55 bounding_box: Tuple[int, int, int, int, int, int], 56 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 57 download: bool = False, 58) -> str: 59 """Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store. 60 61 Args: 62 path: Filepath to a folder where the cached zarr store will be saved. 63 bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel 64 coordinates, at the dataset's native resolution (see `SAMPLES`). 65 sample: Which neuromast to use. See `SAMPLES` for the available keys. 66 download: Whether to stream and cache the data if it is not present. 67 68 Returns: 69 The filepath to the cached zarr store. 70 """ 71 import zarr 72 from zarr.codecs import BloscCodec 73 74 if sample not in SAMPLES: 75 raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}") 76 77 os.makedirs(str(path), exist_ok=True) 78 zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr") 79 80 root = zarr.open_group(zarr_path, mode="a") 81 if "raw" in root and "labels" in root: 82 return zarr_path 83 84 if not download: 85 raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.") 86 87 z_min, z_max, y_min, y_max, x_min, x_max = bounding_box 88 info = SAMPLES[sample] 89 shape_z, shape_y, shape_x = info["shape_zyx"] 90 assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \ 91 f"Bounding box exceeds the {sample} volume {info['shape_zyx']}" 92 93 color = _open_remote_array(info["dataset_id"], "color") 94 seg = _open_remote_array(info["dataset_id"], "segmentation") 95 96 # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x). 97 raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 98 labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 99 100 def _make_array(name, data, shuffle): 101 array = root.create_array( 102 name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype, 103 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 104 ) 105 array[:] = data 106 107 root.attrs["bounding_box"] = list(bounding_box) 108 root.attrs["sample"] = sample 109 root.attrs["resolution_nm"] = list(info["resolution_nm"]) 110 root.attrs["labels_are_exhaustive"] = False 111 112 _make_array("raw", raw, shuffle="shuffle") 113 _make_array("labels", labels, shuffle="bitshuffle") 114 115 return zarr_path 116 117 118def get_neuromast_connectomics_paths( 119 path: Union[os.PathLike, str], 120 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 121 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 122 download: bool = False, 123) -> List[str]: 124 """Get paths to cached NeuromastConnectomics zarr stores. 125 126 Args: 127 path: Filepath to a folder where the cached zarr stores will be saved. 128 bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max). 129 sample: Which neuromast to use. See `SAMPLES` for the available keys. 130 download: Whether to stream and cache the data if it is not present. 131 132 Returns: 133 List of filepaths to the cached zarr stores. 134 """ 135 return [get_neuromast_connectomics_data(path, bbox, sample, download) for bbox in bounding_boxes] 136 137 138def get_neuromast_connectomics_dataset( 139 path: Union[os.PathLike, str], 140 patch_shape: Tuple[int, int, int], 141 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 142 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 143 download: bool = False, 144 **kwargs, 145) -> Dataset: 146 """Get the NeuromastConnectomics dataset for cell instance segmentation. 147 148 Args: 149 path: Filepath to a folder where the cached zarr stores will be saved. 150 patch_shape: The patch shape (z, y, x) to use for training. 151 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 152 sample: Which neuromast to use. See `SAMPLES` for the available keys. 153 download: Whether to stream and cache data if not already present. 154 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 155 156 Returns: 157 The segmentation dataset. 158 """ 159 assert len(patch_shape) == 3 160 161 paths = get_neuromast_connectomics_paths(path, bounding_boxes, sample, download) 162 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 163 164 return torch_em.default_segmentation_dataset( 165 raw_paths=paths, 166 raw_key="raw", 167 label_paths=paths, 168 label_key="labels", 169 patch_shape=patch_shape, 170 **kwargs, 171 ) 172 173 174def get_neuromast_connectomics_loader( 175 path: Union[os.PathLike, str], 176 patch_shape: Tuple[int, int, int], 177 batch_size: int, 178 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 179 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 180 download: bool = False, 181 **kwargs, 182) -> DataLoader: 183 """Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset. 184 185 Args: 186 path: Filepath to a folder where the cached zarr stores will be saved. 187 patch_shape: The patch shape (z, y, x) to use for training. 188 batch_size: The batch size for training. 189 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 190 sample: Which neuromast to use. See `SAMPLES` for the available keys. 191 download: Whether to stream and cache data if not already present. 192 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 193 PyTorch DataLoader. 194 195 Returns: 196 The DataLoader. 197 """ 198 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 199 dataset = get_neuromast_connectomics_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs) 200 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
54def get_neuromast_connectomics_data( 55 path: Union[os.PathLike, str], 56 bounding_box: Tuple[int, int, int, int, int, int], 57 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 58 download: bool = False, 59) -> str: 60 """Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store. 61 62 Args: 63 path: Filepath to a folder where the cached zarr store will be saved. 64 bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel 65 coordinates, at the dataset's native resolution (see `SAMPLES`). 66 sample: Which neuromast to use. See `SAMPLES` for the available keys. 67 download: Whether to stream and cache the data if it is not present. 68 69 Returns: 70 The filepath to the cached zarr store. 71 """ 72 import zarr 73 from zarr.codecs import BloscCodec 74 75 if sample not in SAMPLES: 76 raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}") 77 78 os.makedirs(str(path), exist_ok=True) 79 zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr") 80 81 root = zarr.open_group(zarr_path, mode="a") 82 if "raw" in root and "labels" in root: 83 return zarr_path 84 85 if not download: 86 raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.") 87 88 z_min, z_max, y_min, y_max, x_min, x_max = bounding_box 89 info = SAMPLES[sample] 90 shape_z, shape_y, shape_x = info["shape_zyx"] 91 assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \ 92 f"Bounding box exceeds the {sample} volume {info['shape_zyx']}" 93 94 color = _open_remote_array(info["dataset_id"], "color") 95 seg = _open_remote_array(info["dataset_id"], "segmentation") 96 97 # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x). 98 raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 99 labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0)) 100 101 def _make_array(name, data, shuffle): 102 array = root.create_array( 103 name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype, 104 compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle), 105 ) 106 array[:] = data 107 108 root.attrs["bounding_box"] = list(bounding_box) 109 root.attrs["sample"] = sample 110 root.attrs["resolution_nm"] = list(info["resolution_nm"]) 111 root.attrs["labels_are_exhaustive"] = False 112 113 _make_array("raw", raw, shuffle="shuffle") 114 _make_array("labels", labels, shuffle="bitshuffle") 115 116 return zarr_path
Stream a subvolume of the NeuromastConnectomics dataset and cache it as a zarr v3 store.
Arguments:
- path: Filepath to a folder where the cached zarr store will be saved.
- bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
coordinates, at the dataset's native resolution (see
SAMPLES). - sample: Which neuromast to use. See
SAMPLESfor the available keys. - download: Whether to stream and cache the data if it is not present.
Returns:
The filepath to the cached zarr store.
119def get_neuromast_connectomics_paths( 120 path: Union[os.PathLike, str], 121 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 122 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 123 download: bool = False, 124) -> List[str]: 125 """Get paths to cached NeuromastConnectomics zarr stores. 126 127 Args: 128 path: Filepath to a folder where the cached zarr stores will be saved. 129 bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max). 130 sample: Which neuromast to use. See `SAMPLES` for the available keys. 131 download: Whether to stream and cache the data if it is not present. 132 133 Returns: 134 List of filepaths to the cached zarr stores. 135 """ 136 return [get_neuromast_connectomics_data(path, bbox, sample, download) for bbox in bounding_boxes]
Get paths to cached NeuromastConnectomics zarr stores.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- sample: Which neuromast to use. See
SAMPLESfor the available keys. - download: Whether to stream and cache the data if it is not present.
Returns:
List of filepaths to the cached zarr stores.
139def get_neuromast_connectomics_dataset( 140 path: Union[os.PathLike, str], 141 patch_shape: Tuple[int, int, int], 142 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 143 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 144 download: bool = False, 145 **kwargs, 146) -> Dataset: 147 """Get the NeuromastConnectomics dataset for cell instance segmentation. 148 149 Args: 150 path: Filepath to a folder where the cached zarr stores will be saved. 151 patch_shape: The patch shape (z, y, x) to use for training. 152 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 153 sample: Which neuromast to use. See `SAMPLES` for the available keys. 154 download: Whether to stream and cache data if not already present. 155 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 156 157 Returns: 158 The segmentation dataset. 159 """ 160 assert len(patch_shape) == 3 161 162 paths = get_neuromast_connectomics_paths(path, bounding_boxes, sample, download) 163 kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True) 164 165 return torch_em.default_segmentation_dataset( 166 raw_paths=paths, 167 raw_key="raw", 168 label_paths=paths, 169 label_key="labels", 170 patch_shape=patch_shape, 171 **kwargs, 172 )
Get the NeuromastConnectomics dataset for cell instance segmentation.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- sample: Which neuromast to use. See
SAMPLESfor the available keys. - download: Whether to stream and cache data if not already present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
175def get_neuromast_connectomics_loader( 176 path: Union[os.PathLike, str], 177 patch_shape: Tuple[int, int, int], 178 batch_size: int, 179 bounding_boxes: List[Tuple[int, int, int, int, int, int]], 180 sample: Literal["wt1", "wt2", "wt6", "wt7", "wt8", "t1", "t3", "t4", "n1", "n2"] = "wt1", 181 download: bool = False, 182 **kwargs, 183) -> DataLoader: 184 """Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset. 185 186 Args: 187 path: Filepath to a folder where the cached zarr stores will be saved. 188 patch_shape: The patch shape (z, y, x) to use for training. 189 batch_size: The batch size for training. 190 bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max). 191 sample: Which neuromast to use. See `SAMPLES` for the available keys. 192 download: Whether to stream and cache data if not already present. 193 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the 194 PyTorch DataLoader. 195 196 Returns: 197 The DataLoader. 198 """ 199 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 200 dataset = get_neuromast_connectomics_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs) 201 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the DataLoader for cell instance segmentation in the NeuromastConnectomics dataset.
Arguments:
- path: Filepath to a folder where the cached zarr stores will be saved.
- patch_shape: The patch shape (z, y, x) to use for training.
- batch_size: The batch size for training.
- bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
- sample: Which neuromast to use. See
SAMPLESfor the available keys. - download: Whether to stream and cache data if not already present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.