torch_em.data.datasets.electron_microscopy.l4_dense_reconstruction

The L4Dense dataset contains a dense electron microscopy connectomic reconstruction of layer 4 of mouse somatosensory (barrel) cortex, with proofread neuron instance segmentation.

Two samples are available: the full reconstructed volume ('full') and a smaller public demo subvolume ('demo').

Segmentation is dense but not uniformly proofread throughout each volume: some regions still show block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches at internal block boundaries.

The data is available via the Helmstaedter Lab's public WEBKNOSSOS instances (organization MPI_Brain_Research); the paper's own data repository at https://l4dense2019.brain.mpg.de was unreachable at the time of writing this module. No license is specified by the data provider. The dataset was published in https://doi.org/10.1126/science.aay3134. Please cite this publication if you use the dataset in your research.

  1"""The L4Dense dataset contains a dense electron microscopy connectomic reconstruction of layer 4
  2of mouse somatosensory (barrel) cortex, with proofread neuron instance segmentation.
  3
  4Two samples are available: the full reconstructed volume ('full') and a smaller public demo
  5subvolume ('demo').
  6
  7Segmentation is dense but not uniformly proofread throughout each volume: some regions still show
  8block-stitching seams from the automated reconstruction, visible as brief instance ID mismatches
  9at internal block boundaries.
 10
 11The data is available via the Helmstaedter Lab's public WEBKNOSSOS instances (organization
 12MPI_Brain_Research); the paper's own data repository at https://l4dense2019.brain.mpg.de was
 13unreachable at the time of writing this module. No license is specified by the data provider.
 14The dataset was published in https://doi.org/10.1126/science.aay3134.
 15Please cite this publication if you use the dataset in your research.
 16"""
 17
 18import os
 19from typing import Dict, List, Literal, Tuple, Union
 20
 21import numpy as np
 22
 23from torch.utils.data import DataLoader, Dataset
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30SAMPLES: Dict[str, dict] = {
 31    "full": {
 32        "datastore_url": "https://demo.wk1.connectomics.hpccloud.mpg.de/data/zarr",
 33        "dataset_id": "5da84e9401000018006d210b",
 34        "shape_zyx": (3424, 8534, 5599),
 35        "resolution_nm": (28, 11.24, 11.24),
 36    },
 37    "demo": {
 38        "datastore_url": "https://data-humerus.webknossos.org/data/zarr",
 39        "dataset_id": "5a7c737448d7a73cea3c83a8",
 40        "shape_zyx": (714, 1779, 1779),
 41        "resolution_nm": (28, 11.24, 11.24),
 42    },
 43}
 44
 45
 46def _bbox_to_str(bounding_box):
 47    import hashlib
 48    return hashlib.md5("_".join(str(v) for v in bounding_box).encode()).hexdigest()[:12]
 49
 50
 51def _open_remote_array(datastore_url, dataset_id, layer):
 52    import fsspec
 53    import zarr
 54    return zarr.open(fsspec.get_mapper(f"{datastore_url}/{dataset_id}/{layer}/1-1-1"), mode="r", zarr_format=2)
 55
 56
 57def get_l4_dense_reconstruction_data(
 58    path: Union[os.PathLike, str],
 59    bounding_box: Tuple[int, int, int, int, int, int],
 60    sample: Literal["full", "demo"] = "full",
 61    download: bool = False,
 62) -> str:
 63    """Stream a subvolume of the L4Dense dataset and cache it as a zarr v3 store.
 64
 65    Args:
 66        path: Filepath to a folder where the cached zarr store will be saved.
 67        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 68            coordinates, at the dataset's native resolution (see `SAMPLES`).
 69        sample: Which sample to use, 'full' or 'demo'.
 70        download: Whether to stream and cache the data if it is not present.
 71
 72    Returns:
 73        The filepath to the cached zarr store.
 74    """
 75    import zarr
 76    from zarr.codecs import BloscCodec
 77
 78    if sample not in SAMPLES:
 79        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 80
 81    os.makedirs(str(path), exist_ok=True)
 82    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 83
 84    root = zarr.open_group(zarr_path, mode="a")
 85    if "raw" in root and "labels" in root:
 86        return zarr_path
 87
 88    if not download:
 89        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 90
 91    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 92    info = SAMPLES[sample]
 93    shape_z, shape_y, shape_x = info["shape_zyx"]
 94    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 95        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 96
 97    color = _open_remote_array(info["datastore_url"], info["dataset_id"], "color")
 98    seg = _open_remote_array(info["datastore_url"], info["dataset_id"], "segmentation")
 99
100    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
101    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
102    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
103
104    def _make_array(name, data, shuffle):
105        array = root.create_array(
106            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
107            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
108        )
109        array[:] = data
110
111    root.attrs["bounding_box"] = list(bounding_box)
112    root.attrs["sample"] = sample
113    root.attrs["resolution_nm"] = list(info["resolution_nm"])
114    root.attrs["labels_are_exhaustive"] = False
115
116    _make_array("raw", raw, shuffle="shuffle")
117    _make_array("labels", labels, shuffle="bitshuffle")
118
119    return zarr_path
120
121
122def get_l4_dense_reconstruction_paths(
123    path: Union[os.PathLike, str],
124    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
125    sample: Literal["full", "demo"] = "full",
126    download: bool = False,
127) -> List[str]:
128    """Get paths to cached L4Dense zarr stores.
129
130    Args:
131        path: Filepath to a folder where the cached zarr stores will be saved.
132        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
133        sample: Which sample to use, 'full' or 'demo'.
134        download: Whether to stream and cache the data if it is not present.
135
136    Returns:
137        List of filepaths to the cached zarr stores.
138    """
139    return [get_l4_dense_reconstruction_data(path, bbox, sample, download) for bbox in bounding_boxes]
140
141
142def get_l4_dense_reconstruction_dataset(
143    path: Union[os.PathLike, str],
144    patch_shape: Tuple[int, int, int],
145    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
146    sample: Literal["full", "demo"] = "full",
147    download: bool = False,
148    **kwargs,
149) -> Dataset:
150    """Get the L4Dense dataset for neuron instance segmentation.
151
152    Args:
153        path: Filepath to a folder where the cached zarr stores will be saved.
154        patch_shape: The patch shape (z, y, x) to use for training.
155        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
156        sample: Which sample to use, 'full' or 'demo'.
157        download: Whether to stream and cache data if not already present.
158        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
159
160    Returns:
161        The segmentation dataset.
162    """
163    assert len(patch_shape) == 3
164
165    paths = get_l4_dense_reconstruction_paths(path, bounding_boxes, sample, download)
166    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
167
168    return torch_em.default_segmentation_dataset(
169        raw_paths=paths,
170        raw_key="raw",
171        label_paths=paths,
172        label_key="labels",
173        patch_shape=patch_shape,
174        **kwargs,
175    )
176
177
178def get_l4_dense_reconstruction_loader(
179    path: Union[os.PathLike, str],
180    patch_shape: Tuple[int, int, int],
181    batch_size: int,
182    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
183    sample: Literal["full", "demo"] = "full",
184    download: bool = False,
185    **kwargs,
186) -> DataLoader:
187    """Get the DataLoader for neuron instance segmentation in the L4Dense dataset.
188
189    Args:
190        path: Filepath to a folder where the cached zarr stores will be saved.
191        patch_shape: The patch shape (z, y, x) to use for training.
192        batch_size: The batch size for training.
193        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
194        sample: Which sample to use, 'full' or 'demo'.
195        download: Whether to stream and cache data if not already present.
196        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
197            PyTorch DataLoader.
198
199    Returns:
200        The DataLoader.
201    """
202    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
203    dataset = get_l4_dense_reconstruction_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
204    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
SAMPLES: Dict[str, dict] = {'full': {'datastore_url': 'https://demo.wk1.connectomics.hpccloud.mpg.de/data/zarr', 'dataset_id': '5da84e9401000018006d210b', 'shape_zyx': (3424, 8534, 5599), 'resolution_nm': (28, 11.24, 11.24)}, 'demo': {'datastore_url': 'https://data-humerus.webknossos.org/data/zarr', 'dataset_id': '5a7c737448d7a73cea3c83a8', 'shape_zyx': (714, 1779, 1779), 'resolution_nm': (28, 11.24, 11.24)}}
def get_l4_dense_reconstruction_data( path: Union[os.PathLike, str], bounding_box: Tuple[int, int, int, int, int, int], sample: Literal['full', 'demo'] = 'full', download: bool = False) -> str:
 58def get_l4_dense_reconstruction_data(
 59    path: Union[os.PathLike, str],
 60    bounding_box: Tuple[int, int, int, int, int, int],
 61    sample: Literal["full", "demo"] = "full",
 62    download: bool = False,
 63) -> str:
 64    """Stream a subvolume of the L4Dense dataset and cache it as a zarr v3 store.
 65
 66    Args:
 67        path: Filepath to a folder where the cached zarr store will be saved.
 68        bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel
 69            coordinates, at the dataset's native resolution (see `SAMPLES`).
 70        sample: Which sample to use, 'full' or 'demo'.
 71        download: Whether to stream and cache the data if it is not present.
 72
 73    Returns:
 74        The filepath to the cached zarr store.
 75    """
 76    import zarr
 77    from zarr.codecs import BloscCodec
 78
 79    if sample not in SAMPLES:
 80        raise ValueError(f"sample must be one of {list(SAMPLES)}, got {sample!r}")
 81
 82    os.makedirs(str(path), exist_ok=True)
 83    zarr_path = os.path.join(str(path), f"{sample}_{_bbox_to_str(bounding_box)}.zarr")
 84
 85    root = zarr.open_group(zarr_path, mode="a")
 86    if "raw" in root and "labels" in root:
 87        return zarr_path
 88
 89    if not download:
 90        raise RuntimeError(f"No cached data found at '{zarr_path}'. Set download=True to stream it.")
 91
 92    z_min, z_max, y_min, y_max, x_min, x_max = bounding_box
 93    info = SAMPLES[sample]
 94    shape_z, shape_y, shape_x = info["shape_zyx"]
 95    assert z_max <= shape_z and y_max <= shape_y and x_max <= shape_x, \
 96        f"Bounding box exceeds the {sample} volume {info['shape_zyx']}"
 97
 98    color = _open_remote_array(info["datastore_url"], info["dataset_id"], "color")
 99    seg = _open_remote_array(info["datastore_url"], info["dataset_id"], "segmentation")
100
101    # the remote arrays are indexed as (channel, x, y, z); torch_em volumes use (z, y, x).
102    raw = np.transpose(color[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
103    labels = np.transpose(seg[0, x_min:x_max, y_min:y_max, z_min:z_max], (2, 1, 0))
104
105    def _make_array(name, data, shuffle):
106        array = root.create_array(
107            name, shape=data.shape, chunks=(32, 512, 512), dtype=data.dtype,
108            compressors=BloscCodec(cname="zstd", clevel=6, shuffle=shuffle),
109        )
110        array[:] = data
111
112    root.attrs["bounding_box"] = list(bounding_box)
113    root.attrs["sample"] = sample
114    root.attrs["resolution_nm"] = list(info["resolution_nm"])
115    root.attrs["labels_are_exhaustive"] = False
116
117    _make_array("raw", raw, shuffle="shuffle")
118    _make_array("labels", labels, shuffle="bitshuffle")
119
120    return zarr_path

Stream a subvolume of the L4Dense dataset and cache it as a zarr v3 store.

Arguments:
  • path: Filepath to a folder where the cached zarr store will be saved.
  • bounding_box: The region to fetch as (z_min, z_max, y_min, y_max, x_min, x_max) in voxel coordinates, at the dataset's native resolution (see SAMPLES).
  • sample: Which sample to use, 'full' or 'demo'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

The filepath to the cached zarr store.

def get_l4_dense_reconstruction_paths( path: Union[os.PathLike, str], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['full', 'demo'] = 'full', download: bool = False) -> List[str]:
123def get_l4_dense_reconstruction_paths(
124    path: Union[os.PathLike, str],
125    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
126    sample: Literal["full", "demo"] = "full",
127    download: bool = False,
128) -> List[str]:
129    """Get paths to cached L4Dense zarr stores.
130
131    Args:
132        path: Filepath to a folder where the cached zarr stores will be saved.
133        bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
134        sample: Which sample to use, 'full' or 'demo'.
135        download: Whether to stream and cache the data if it is not present.
136
137    Returns:
138        List of filepaths to the cached zarr stores.
139    """
140    return [get_l4_dense_reconstruction_data(path, bbox, sample, download) for bbox in bounding_boxes]

Get paths to cached L4Dense zarr stores.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • bounding_boxes: List of regions to fetch, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which sample to use, 'full' or 'demo'.
  • download: Whether to stream and cache the data if it is not present.
Returns:

List of filepaths to the cached zarr stores.

def get_l4_dense_reconstruction_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['full', 'demo'] = 'full', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
143def get_l4_dense_reconstruction_dataset(
144    path: Union[os.PathLike, str],
145    patch_shape: Tuple[int, int, int],
146    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
147    sample: Literal["full", "demo"] = "full",
148    download: bool = False,
149    **kwargs,
150) -> Dataset:
151    """Get the L4Dense dataset for neuron instance segmentation.
152
153    Args:
154        path: Filepath to a folder where the cached zarr stores will be saved.
155        patch_shape: The patch shape (z, y, x) to use for training.
156        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
157        sample: Which sample to use, 'full' or 'demo'.
158        download: Whether to stream and cache data if not already present.
159        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
160
161    Returns:
162        The segmentation dataset.
163    """
164    assert len(patch_shape) == 3
165
166    paths = get_l4_dense_reconstruction_paths(path, bounding_boxes, sample, download)
167    kwargs = util.update_kwargs(kwargs, "is_seg_dataset", True)
168
169    return torch_em.default_segmentation_dataset(
170        raw_paths=paths,
171        raw_key="raw",
172        label_paths=paths,
173        label_key="labels",
174        patch_shape=patch_shape,
175        **kwargs,
176    )

Get the L4Dense dataset for neuron instance segmentation.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which sample to use, 'full' or 'demo'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_l4_dense_reconstruction_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], batch_size: int, bounding_boxes: List[Tuple[int, int, int, int, int, int]], sample: Literal['full', 'demo'] = 'full', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
179def get_l4_dense_reconstruction_loader(
180    path: Union[os.PathLike, str],
181    patch_shape: Tuple[int, int, int],
182    batch_size: int,
183    bounding_boxes: List[Tuple[int, int, int, int, int, int]],
184    sample: Literal["full", "demo"] = "full",
185    download: bool = False,
186    **kwargs,
187) -> DataLoader:
188    """Get the DataLoader for neuron instance segmentation in the L4Dense dataset.
189
190    Args:
191        path: Filepath to a folder where the cached zarr stores will be saved.
192        patch_shape: The patch shape (z, y, x) to use for training.
193        batch_size: The batch size for training.
194        bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
195        sample: Which sample to use, 'full' or 'demo'.
196        download: Whether to stream and cache data if not already present.
197        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the
198            PyTorch DataLoader.
199
200    Returns:
201        The DataLoader.
202    """
203    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
204    dataset = get_l4_dense_reconstruction_dataset(path, patch_shape, bounding_boxes, sample, download, **ds_kwargs)
205    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DataLoader for neuron instance segmentation in the L4Dense dataset.

Arguments:
  • path: Filepath to a folder where the cached zarr stores will be saved.
  • patch_shape: The patch shape (z, y, x) to use for training.
  • batch_size: The batch size for training.
  • bounding_boxes: List of subvolumes to use, each as (z_min, z_max, y_min, y_max, x_min, x_max).
  • sample: Which sample to use, 'full' or 'demo'.
  • download: Whether to stream and cache data if not already present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.