torch_em.data.datasets.medical.hippo_subfields

The Hippo-Subfields dataset contains annotations for hippocampal subfield segmentation in 7 Tesla brain MRI.

The dataset consists of paired whole-brain T1-weighted and T2-weighted MRI scans acquired at 3 Tesla and 7 Tesla from 20 healthy volunteers. The hippocampal formation subfields were manually delineated bilaterally on every coronal section of the 7 Tesla T2-weighted scans (upsampled to 0.7 mm slice thickness), which is the modality and annotation used by this module.

NOTE: The left and right hippocampus are distributed as two separate label volumes per subject, both on the same whole-brain grid as the raw scan and with non-overlapping foreground regions. This module merges them (via a voxel-wise maximum) into a single semantic label volume with the following 7 foreground classes, following LABEL_IDS: 0 = background, 1 = subiculum (SUB), 2 = CA2, 3 = CA1, 4 = CA4 and dentate gyrus (CA4&DG), 5 = entorhinal cortex (ERC), 6 = CA3, 7 = hippocampal tail.

The dataset is located at https://doi.org/10.25452/figshare.plus.26075713.v1 and is distributed under the CC BY 4.0 license.

This dataset is from the publication https://doi.org/10.1038/s41597-025-04586-9. Please cite it if you use this dataset in your research.

  1"""The Hippo-Subfields dataset contains annotations for hippocampal subfield segmentation in 7 Tesla
  2brain MRI.
  3
  4The dataset consists of paired whole-brain T1-weighted and T2-weighted MRI scans acquired at 3 Tesla
  5and 7 Tesla from 20 healthy volunteers. The hippocampal formation subfields were manually delineated
  6bilaterally on every coronal section of the 7 Tesla T2-weighted scans (upsampled to 0.7 mm slice
  7thickness), which is the modality and annotation used by this module.
  8
  9NOTE: The left and right hippocampus are distributed as two separate label volumes per subject, both
 10on the same whole-brain grid as the raw scan and with non-overlapping foreground regions. This module
 11merges them (via a voxel-wise maximum) into a single semantic label volume with the following 7
 12foreground classes, following `LABEL_IDS`:
 130 = background, 1 = subiculum (SUB), 2 = CA2, 3 = CA1, 4 = CA4 and dentate gyrus (CA4&DG),
 145 = entorhinal cortex (ERC), 6 = CA3, 7 = hippocampal tail.
 15
 16The dataset is located at https://doi.org/10.25452/figshare.plus.26075713.v1 and is distributed under
 17the CC BY 4.0 license.
 18
 19This dataset is from the publication https://doi.org/10.1038/s41597-025-04586-9.
 20Please cite it if you use this dataset in your research.
 21"""
 22
 23import os
 24from glob import glob
 25from tqdm import tqdm
 26from natsort import natsorted
 27from typing import Union, Tuple, List
 28
 29import numpy as np
 30
 31from torch.utils.data import Dataset, DataLoader
 32
 33import torch_em
 34
 35from .. import util
 36
 37
 38URL = "https://ndownloader.figshare.com/files/50052801"
 39CHECKSUM = "2b2d32c6779cfdb79638f4de34934d62484f2eabbcbb17d43732f3b96c14d54c"
 40
 41LABEL_IDS = {"background": 0, "SUB": 1, "CA2": 2, "CA1": 3, "CA4&DG": 4, "ERC": 5, "CA3": 6, "tail": 7}
 42"""The semantic label ids of the hippocampal subfield classes."""
 43
 44
 45def _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir):
 46    import h5py
 47    import nibabel as nib
 48
 49    os.makedirs(preprocessed_dir, exist_ok=True)
 50
 51    label_paths = natsorted(glob(os.path.join(label_dir, "sub-*-L.nii.gz")))
 52    subject_ids = [os.path.basename(p).split("-")[1] for p in label_paths]
 53
 54    for subject_id in tqdm(subject_ids, desc="Preprocessing the Hippo-Subfields data"):
 55        out_path = os.path.join(preprocessed_dir, f"sub-{subject_id}.h5")
 56        if os.path.exists(out_path):
 57            continue
 58
 59        raw_path = os.path.join(raw_dir, f"t2_s{subject_id}_0.7.nii")
 60        left_path = os.path.join(label_dir, f"sub-{subject_id}-L.nii.gz")
 61        right_path = os.path.join(label_dir, f"sub-{subject_id}-R.nii.gz")
 62
 63        raw = np.asarray(nib.load(raw_path).dataobj).astype("float32")
 64        left = np.asarray(nib.load(left_path).dataobj)
 65        right = np.asarray(nib.load(right_path).dataobj)
 66        labels = np.maximum(left, right).astype("uint8")
 67        assert raw.shape == labels.shape, f"Shape mismatch for {subject_id}: {raw.shape} vs. {labels.shape}."
 68
 69        with h5py.File(f"{out_path}.tmp", "w") as f:
 70            f.create_dataset("raw", data=raw, compression="gzip")
 71            f.create_dataset("labels", data=labels, compression="gzip")
 72        os.rename(f"{out_path}.tmp", out_path)
 73
 74    return preprocessed_dir
 75
 76
 77def get_hippo_subfields_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 78    """Download the Hippo-Subfields dataset.
 79
 80    Args:
 81        path: Filepath to a folder where the data is downloaded for further processing.
 82        download: Whether to download the data if it is not present.
 83
 84    Returns:
 85        Filepath where the data is preprocessed.
 86    """
 87    preprocessed_dir = os.path.join(path, "preprocessed")
 88    if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0:
 89        return preprocessed_dir
 90
 91    os.makedirs(path, exist_ok=True)
 92
 93    data_dir = os.path.join(path, "hippo_subfield")
 94    if not os.path.exists(data_dir):
 95        zip_path = os.path.join(path, "hippo_subfield.zip")
 96        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 97        util.unzip(zip_path=zip_path, dst=path)
 98
 99    raw_dir = os.path.join(data_dir, "7T_T2w_0.7_for_subfield_delineation")
100    label_dir = os.path.join(data_dir, "hippo_label")
101    return _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir)
102
103
104def get_hippo_subfields_paths(
105    path: Union[os.PathLike, str], download: bool = False
106) -> Tuple[List[str], List[str]]:
107    """Get paths to the Hippo-Subfields data.
108
109    Args:
110        path: Filepath to a folder where the data is downloaded for further processing.
111        download: Whether to download the data if it is not present.
112
113    Returns:
114        List of filepaths for the image data.
115        List of filepaths for the label data.
116    """
117    preprocessed_dir = get_hippo_subfields_data(path, download)
118    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
119    assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'."
120    return volume_paths, volume_paths
121
122
123def get_hippo_subfields_dataset(
124    path: Union[os.PathLike, str],
125    patch_shape: Tuple[int, ...],
126    resize_inputs: bool = False,
127    download: bool = False,
128    **kwargs
129) -> Dataset:
130    """Get the Hippo-Subfields dataset for hippocampal subfield segmentation.
131
132    Args:
133        path: Filepath to a folder where the data is downloaded for further processing.
134        patch_shape: The patch shape to use for training.
135        resize_inputs: Whether to resize inputs to the desired patch shape.
136        download: Whether to download the data if it is not present.
137        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
138
139    Returns:
140        The segmentation dataset.
141    """
142    raw_paths, label_paths = get_hippo_subfields_paths(path, download)
143
144    if resize_inputs:
145        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
146        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
147            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
148        )
149
150    return torch_em.default_segmentation_dataset(
151        raw_paths=raw_paths,
152        raw_key="raw",
153        label_paths=label_paths,
154        label_key="labels",
155        patch_shape=patch_shape,
156        is_seg_dataset=True,
157        **kwargs
158    )
159
160
161def get_hippo_subfields_loader(
162    path: Union[os.PathLike, str],
163    batch_size: int,
164    patch_shape: Tuple[int, ...],
165    resize_inputs: bool = False,
166    download: bool = False,
167    **kwargs
168) -> DataLoader:
169    """Get the Hippo-Subfields dataloader for hippocampal subfield segmentation.
170
171    Args:
172        path: Filepath to a folder where the data is downloaded for further processing.
173        batch_size: The batch size for training.
174        patch_shape: The patch shape to use for training.
175        resize_inputs: Whether to resize inputs to the desired patch shape.
176        download: Whether to download the data if it is not present.
177        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
178
179    Returns:
180        The DataLoader.
181    """
182    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
183    dataset = get_hippo_subfields_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
184    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://ndownloader.figshare.com/files/50052801'
CHECKSUM = '2b2d32c6779cfdb79638f4de34934d62484f2eabbcbb17d43732f3b96c14d54c'
LABEL_IDS = {'background': 0, 'SUB': 1, 'CA2': 2, 'CA1': 3, 'CA4&DG': 4, 'ERC': 5, 'CA3': 6, 'tail': 7}

The semantic label ids of the hippocampal subfield classes.

def get_hippo_subfields_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 78def get_hippo_subfields_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 79    """Download the Hippo-Subfields dataset.
 80
 81    Args:
 82        path: Filepath to a folder where the data is downloaded for further processing.
 83        download: Whether to download the data if it is not present.
 84
 85    Returns:
 86        Filepath where the data is preprocessed.
 87    """
 88    preprocessed_dir = os.path.join(path, "preprocessed")
 89    if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0:
 90        return preprocessed_dir
 91
 92    os.makedirs(path, exist_ok=True)
 93
 94    data_dir = os.path.join(path, "hippo_subfield")
 95    if not os.path.exists(data_dir):
 96        zip_path = os.path.join(path, "hippo_subfield.zip")
 97        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 98        util.unzip(zip_path=zip_path, dst=path)
 99
100    raw_dir = os.path.join(data_dir, "7T_T2w_0.7_for_subfield_delineation")
101    label_dir = os.path.join(data_dir, "hippo_label")
102    return _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir)

Download the Hippo-Subfields dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is preprocessed.

def get_hippo_subfields_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
105def get_hippo_subfields_paths(
106    path: Union[os.PathLike, str], download: bool = False
107) -> Tuple[List[str], List[str]]:
108    """Get paths to the Hippo-Subfields data.
109
110    Args:
111        path: Filepath to a folder where the data is downloaded for further processing.
112        download: Whether to download the data if it is not present.
113
114    Returns:
115        List of filepaths for the image data.
116        List of filepaths for the label data.
117    """
118    preprocessed_dir = get_hippo_subfields_data(path, download)
119    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
120    assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'."
121    return volume_paths, volume_paths

Get paths to the Hippo-Subfields data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_hippo_subfields_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
124def get_hippo_subfields_dataset(
125    path: Union[os.PathLike, str],
126    patch_shape: Tuple[int, ...],
127    resize_inputs: bool = False,
128    download: bool = False,
129    **kwargs
130) -> Dataset:
131    """Get the Hippo-Subfields dataset for hippocampal subfield segmentation.
132
133    Args:
134        path: Filepath to a folder where the data is downloaded for further processing.
135        patch_shape: The patch shape to use for training.
136        resize_inputs: Whether to resize inputs to the desired patch shape.
137        download: Whether to download the data if it is not present.
138        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
139
140    Returns:
141        The segmentation dataset.
142    """
143    raw_paths, label_paths = get_hippo_subfields_paths(path, download)
144
145    if resize_inputs:
146        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
147        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
148            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
149        )
150
151    return torch_em.default_segmentation_dataset(
152        raw_paths=raw_paths,
153        raw_key="raw",
154        label_paths=label_paths,
155        label_key="labels",
156        patch_shape=patch_shape,
157        is_seg_dataset=True,
158        **kwargs
159    )

Get the Hippo-Subfields dataset for hippocampal subfield segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_hippo_subfields_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
162def get_hippo_subfields_loader(
163    path: Union[os.PathLike, str],
164    batch_size: int,
165    patch_shape: Tuple[int, ...],
166    resize_inputs: bool = False,
167    download: bool = False,
168    **kwargs
169) -> DataLoader:
170    """Get the Hippo-Subfields dataloader for hippocampal subfield segmentation.
171
172    Args:
173        path: Filepath to a folder where the data is downloaded for further processing.
174        batch_size: The batch size for training.
175        patch_shape: The patch shape to use for training.
176        resize_inputs: Whether to resize inputs to the desired patch shape.
177        download: Whether to download the data if it is not present.
178        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
179
180    Returns:
181        The DataLoader.
182    """
183    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
184    dataset = get_hippo_subfields_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
185    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Hippo-Subfields dataloader for hippocampal subfield segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.