torch_em.data.datasets.medical.hippo_subfields
The Hippo-Subfields dataset contains annotations for hippocampal subfield segmentation in 7 Tesla brain MRI.
The dataset consists of paired whole-brain T1-weighted and T2-weighted MRI scans acquired at 3 Tesla and 7 Tesla from 20 healthy volunteers. The hippocampal formation subfields were manually delineated bilaterally on every coronal section of the 7 Tesla T2-weighted scans (upsampled to 0.7 mm slice thickness), which is the modality and annotation used by this module.
NOTE: The left and right hippocampus are distributed as two separate label volumes per subject, both
on the same whole-brain grid as the raw scan and with non-overlapping foreground regions. This module
merges them (via a voxel-wise maximum) into a single semantic label volume with the following 7
foreground classes, following LABEL_IDS:
0 = background, 1 = subiculum (SUB), 2 = CA2, 3 = CA1, 4 = CA4 and dentate gyrus (CA4&DG),
5 = entorhinal cortex (ERC), 6 = CA3, 7 = hippocampal tail.
The dataset is located at https://doi.org/10.25452/figshare.plus.26075713.v1 and is distributed under the CC BY 4.0 license.
This dataset is from the publication https://doi.org/10.1038/s41597-025-04586-9. Please cite it if you use this dataset in your research.
1"""The Hippo-Subfields dataset contains annotations for hippocampal subfield segmentation in 7 Tesla 2brain MRI. 3 4The dataset consists of paired whole-brain T1-weighted and T2-weighted MRI scans acquired at 3 Tesla 5and 7 Tesla from 20 healthy volunteers. The hippocampal formation subfields were manually delineated 6bilaterally on every coronal section of the 7 Tesla T2-weighted scans (upsampled to 0.7 mm slice 7thickness), which is the modality and annotation used by this module. 8 9NOTE: The left and right hippocampus are distributed as two separate label volumes per subject, both 10on the same whole-brain grid as the raw scan and with non-overlapping foreground regions. This module 11merges them (via a voxel-wise maximum) into a single semantic label volume with the following 7 12foreground classes, following `LABEL_IDS`: 130 = background, 1 = subiculum (SUB), 2 = CA2, 3 = CA1, 4 = CA4 and dentate gyrus (CA4&DG), 145 = entorhinal cortex (ERC), 6 = CA3, 7 = hippocampal tail. 15 16The dataset is located at https://doi.org/10.25452/figshare.plus.26075713.v1 and is distributed under 17the CC BY 4.0 license. 18 19This dataset is from the publication https://doi.org/10.1038/s41597-025-04586-9. 20Please cite it if you use this dataset in your research. 21""" 22 23import os 24from glob import glob 25from tqdm import tqdm 26from natsort import natsorted 27from typing import Union, Tuple, List 28 29import numpy as np 30 31from torch.utils.data import Dataset, DataLoader 32 33import torch_em 34 35from .. import util 36 37 38URL = "https://ndownloader.figshare.com/files/50052801" 39CHECKSUM = "2b2d32c6779cfdb79638f4de34934d62484f2eabbcbb17d43732f3b96c14d54c" 40 41LABEL_IDS = {"background": 0, "SUB": 1, "CA2": 2, "CA1": 3, "CA4&DG": 4, "ERC": 5, "CA3": 6, "tail": 7} 42"""The semantic label ids of the hippocampal subfield classes.""" 43 44 45def _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir): 46 import h5py 47 import nibabel as nib 48 49 os.makedirs(preprocessed_dir, exist_ok=True) 50 51 label_paths = natsorted(glob(os.path.join(label_dir, "sub-*-L.nii.gz"))) 52 subject_ids = [os.path.basename(p).split("-")[1] for p in label_paths] 53 54 for subject_id in tqdm(subject_ids, desc="Preprocessing the Hippo-Subfields data"): 55 out_path = os.path.join(preprocessed_dir, f"sub-{subject_id}.h5") 56 if os.path.exists(out_path): 57 continue 58 59 raw_path = os.path.join(raw_dir, f"t2_s{subject_id}_0.7.nii") 60 left_path = os.path.join(label_dir, f"sub-{subject_id}-L.nii.gz") 61 right_path = os.path.join(label_dir, f"sub-{subject_id}-R.nii.gz") 62 63 raw = np.asarray(nib.load(raw_path).dataobj).astype("float32") 64 left = np.asarray(nib.load(left_path).dataobj) 65 right = np.asarray(nib.load(right_path).dataobj) 66 labels = np.maximum(left, right).astype("uint8") 67 assert raw.shape == labels.shape, f"Shape mismatch for {subject_id}: {raw.shape} vs. {labels.shape}." 68 69 with h5py.File(f"{out_path}.tmp", "w") as f: 70 f.create_dataset("raw", data=raw, compression="gzip") 71 f.create_dataset("labels", data=labels, compression="gzip") 72 os.rename(f"{out_path}.tmp", out_path) 73 74 return preprocessed_dir 75 76 77def get_hippo_subfields_data(path: Union[os.PathLike, str], download: bool = False) -> str: 78 """Download the Hippo-Subfields dataset. 79 80 Args: 81 path: Filepath to a folder where the data is downloaded for further processing. 82 download: Whether to download the data if it is not present. 83 84 Returns: 85 Filepath where the data is preprocessed. 86 """ 87 preprocessed_dir = os.path.join(path, "preprocessed") 88 if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0: 89 return preprocessed_dir 90 91 os.makedirs(path, exist_ok=True) 92 93 data_dir = os.path.join(path, "hippo_subfield") 94 if not os.path.exists(data_dir): 95 zip_path = os.path.join(path, "hippo_subfield.zip") 96 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 97 util.unzip(zip_path=zip_path, dst=path) 98 99 raw_dir = os.path.join(data_dir, "7T_T2w_0.7_for_subfield_delineation") 100 label_dir = os.path.join(data_dir, "hippo_label") 101 return _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir) 102 103 104def get_hippo_subfields_paths( 105 path: Union[os.PathLike, str], download: bool = False 106) -> Tuple[List[str], List[str]]: 107 """Get paths to the Hippo-Subfields data. 108 109 Args: 110 path: Filepath to a folder where the data is downloaded for further processing. 111 download: Whether to download the data if it is not present. 112 113 Returns: 114 List of filepaths for the image data. 115 List of filepaths for the label data. 116 """ 117 preprocessed_dir = get_hippo_subfields_data(path, download) 118 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 119 assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'." 120 return volume_paths, volume_paths 121 122 123def get_hippo_subfields_dataset( 124 path: Union[os.PathLike, str], 125 patch_shape: Tuple[int, ...], 126 resize_inputs: bool = False, 127 download: bool = False, 128 **kwargs 129) -> Dataset: 130 """Get the Hippo-Subfields dataset for hippocampal subfield segmentation. 131 132 Args: 133 path: Filepath to a folder where the data is downloaded for further processing. 134 patch_shape: The patch shape to use for training. 135 resize_inputs: Whether to resize inputs to the desired patch shape. 136 download: Whether to download the data if it is not present. 137 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 138 139 Returns: 140 The segmentation dataset. 141 """ 142 raw_paths, label_paths = get_hippo_subfields_paths(path, download) 143 144 if resize_inputs: 145 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 146 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 147 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 148 ) 149 150 return torch_em.default_segmentation_dataset( 151 raw_paths=raw_paths, 152 raw_key="raw", 153 label_paths=label_paths, 154 label_key="labels", 155 patch_shape=patch_shape, 156 is_seg_dataset=True, 157 **kwargs 158 ) 159 160 161def get_hippo_subfields_loader( 162 path: Union[os.PathLike, str], 163 batch_size: int, 164 patch_shape: Tuple[int, ...], 165 resize_inputs: bool = False, 166 download: bool = False, 167 **kwargs 168) -> DataLoader: 169 """Get the Hippo-Subfields dataloader for hippocampal subfield segmentation. 170 171 Args: 172 path: Filepath to a folder where the data is downloaded for further processing. 173 batch_size: The batch size for training. 174 patch_shape: The patch shape to use for training. 175 resize_inputs: Whether to resize inputs to the desired patch shape. 176 download: Whether to download the data if it is not present. 177 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 178 179 Returns: 180 The DataLoader. 181 """ 182 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 183 dataset = get_hippo_subfields_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 184 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
The semantic label ids of the hippocampal subfield classes.
78def get_hippo_subfields_data(path: Union[os.PathLike, str], download: bool = False) -> str: 79 """Download the Hippo-Subfields dataset. 80 81 Args: 82 path: Filepath to a folder where the data is downloaded for further processing. 83 download: Whether to download the data if it is not present. 84 85 Returns: 86 Filepath where the data is preprocessed. 87 """ 88 preprocessed_dir = os.path.join(path, "preprocessed") 89 if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0: 90 return preprocessed_dir 91 92 os.makedirs(path, exist_ok=True) 93 94 data_dir = os.path.join(path, "hippo_subfield") 95 if not os.path.exists(data_dir): 96 zip_path = os.path.join(path, "hippo_subfield.zip") 97 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 98 util.unzip(zip_path=zip_path, dst=path) 99 100 raw_dir = os.path.join(data_dir, "7T_T2w_0.7_for_subfield_delineation") 101 label_dir = os.path.join(data_dir, "hippo_label") 102 return _preprocess_hippo_subfields(raw_dir, label_dir, preprocessed_dir)
Download the Hippo-Subfields dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is preprocessed.
105def get_hippo_subfields_paths( 106 path: Union[os.PathLike, str], download: bool = False 107) -> Tuple[List[str], List[str]]: 108 """Get paths to the Hippo-Subfields data. 109 110 Args: 111 path: Filepath to a folder where the data is downloaded for further processing. 112 download: Whether to download the data if it is not present. 113 114 Returns: 115 List of filepaths for the image data. 116 List of filepaths for the label data. 117 """ 118 preprocessed_dir = get_hippo_subfields_data(path, download) 119 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 120 assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'." 121 return volume_paths, volume_paths
Get paths to the Hippo-Subfields data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
124def get_hippo_subfields_dataset( 125 path: Union[os.PathLike, str], 126 patch_shape: Tuple[int, ...], 127 resize_inputs: bool = False, 128 download: bool = False, 129 **kwargs 130) -> Dataset: 131 """Get the Hippo-Subfields dataset for hippocampal subfield segmentation. 132 133 Args: 134 path: Filepath to a folder where the data is downloaded for further processing. 135 patch_shape: The patch shape to use for training. 136 resize_inputs: Whether to resize inputs to the desired patch shape. 137 download: Whether to download the data if it is not present. 138 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 139 140 Returns: 141 The segmentation dataset. 142 """ 143 raw_paths, label_paths = get_hippo_subfields_paths(path, download) 144 145 if resize_inputs: 146 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 147 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 148 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 149 ) 150 151 return torch_em.default_segmentation_dataset( 152 raw_paths=raw_paths, 153 raw_key="raw", 154 label_paths=label_paths, 155 label_key="labels", 156 patch_shape=patch_shape, 157 is_seg_dataset=True, 158 **kwargs 159 )
Get the Hippo-Subfields dataset for hippocampal subfield segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
162def get_hippo_subfields_loader( 163 path: Union[os.PathLike, str], 164 batch_size: int, 165 patch_shape: Tuple[int, ...], 166 resize_inputs: bool = False, 167 download: bool = False, 168 **kwargs 169) -> DataLoader: 170 """Get the Hippo-Subfields dataloader for hippocampal subfield segmentation. 171 172 Args: 173 path: Filepath to a folder where the data is downloaded for further processing. 174 batch_size: The batch size for training. 175 patch_shape: The patch shape to use for training. 176 resize_inputs: Whether to resize inputs to the desired patch shape. 177 download: Whether to download the data if it is not present. 178 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 179 180 Returns: 181 The DataLoader. 182 """ 183 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 184 dataset = get_hippo_subfields_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 185 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Hippo-Subfields dataloader for hippocampal subfield segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.