torch_em.data.datasets.medical.longciu
The LongCIU dataset contains annotations for ground glass opacity and consolidation in chest CT scans of Long COVID patients.
This dataset is located at https://doi.org/10.25820/data.007301, and is released under the Open Data Commons Attribution License (ODC-By) v1.0. The dataset is from the publication https://doi.org/10.1038/s41597-025-04709-2. Please cite it if you use this dataset for your research.
1"""The LongCIU dataset contains annotations for ground glass opacity and consolidation 2in chest CT scans of Long COVID patients. 3 4This dataset is located at https://doi.org/10.25820/data.007301, and is released 5under the Open Data Commons Attribution License (ODC-By) v1.0. 6The dataset is from the publication https://doi.org/10.1038/s41597-025-04709-2. 7Please cite it if you use this dataset for your research. 8""" 9 10import os 11import json 12from glob import glob 13from natsort import natsorted 14from typing import Tuple, Union, Literal, List 15 16import numpy as np 17 18from torch.utils.data import Dataset, DataLoader 19 20import torch_em 21 22from .. import util 23 24 25BASE_URL = "https://iro.uiowa.edu/view/fileRedirect?instCode=01IOWA_INST&download=true&filePid={}" 26 27FILE_PIDS = { 28 "img": "13931395260002771", 29 "staple": "13931395270002771", 30 "1": "13931395300002771", 31 "2": "13931395290002771", 32 "3": "13931395280002771", 33 "splits": "13931395240002771", 34} 35 36FNAMES = { 37 "img": "longciu_img.nii.gz", 38 "staple": "longciu_STAPLE_tgt.nii.gz", 39 "1": "longciu_1_tgt.nii.gz", 40 "2": "longciu_2_tgt.nii.gz", 41 "3": "longciu_3_tgt.nii.gz", 42 "splits": "longciu_splits.json", 43} 44 45CHECKSUMS = { 46 "img": "af54aff713e7054d204b5e5e2aa0598c856006229901eea791ece8ae2efe8ab8", 47 "staple": "6f0d64315c5010b296285f81a54adec0c6eae5fc643c4e21f4546129d743ea13", 48 "1": "f9cd0e605f6fcd701a64a262ba15d43a46fe2f1b62a4045e83195e4962f022ed", 49 "2": "4e54ec5add75b2ee48f34be6bd25c74f23d550a96430ff10c04e711383f9647f", 50 "3": "aee04feddb18cbbe717418ce8b6838b58fc415f0b5b13b97b1c682faed1bb7cd", 51 "splits": "7f4d62f33f68a16cd2df10cc320e087d43802970de63e6e459885f5c7e5604bf", 52} 53 54 55def _preprocess_data(path): 56 import h5py 57 import nibabel as nib 58 59 def _load(name): 60 vol = nib.load(os.path.join(path, FNAMES[name])).get_fdata() 61 return vol.transpose(2, 0, 1) 62 63 raw = _load("img").astype("float32") 64 labels = {rater: np.rint(_load(rater)).astype("uint8") for rater in ["staple", "1", "2", "3"]} 65 66 with open(os.path.join(path, FNAMES["splits"])) as f: 67 splits = json.load(f) 68 69 data_dir = os.path.join(path, "data") 70 os.makedirs(data_dir, exist_ok=True) 71 72 for split, ids in splits.items(): 73 ids = sorted(ids) 74 with h5py.File(os.path.join(data_dir, f"longciu_{split}.h5"), "w") as f: 75 f.create_dataset("raw", data=raw[ids], compression="gzip") 76 for rater, vol in labels.items(): 77 f.create_dataset(f"labels/{rater}", data=vol[ids], compression="gzip") 78 79 for name in ["img", "staple", "1", "2", "3"]: 80 os.remove(os.path.join(path, FNAMES[name])) 81 82 83def get_longciu_data(path: Union[os.PathLike, str], download: bool = False) -> str: 84 """Download the LongCIU dataset. 85 86 Args: 87 path: Filepath to a folder where the data is downloaded for further processing. 88 download: Whether to download the data if it is not present. 89 90 Returns: 91 Filepath where the preprocessed data is stored. 92 """ 93 data_dir = os.path.join(path, "data") 94 if os.path.exists(data_dir): 95 return data_dir 96 97 os.makedirs(path, exist_ok=True) 98 99 for name, pid in FILE_PIDS.items(): 100 fpath = os.path.join(path, FNAMES[name]) 101 util.download_source(path=fpath, url=BASE_URL.format(pid), download=download, checksum=CHECKSUMS[name]) 102 103 _preprocess_data(path) 104 105 return data_dir 106 107 108def get_longciu_paths( 109 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False 110) -> List[str]: 111 """Get paths to the LongCIU data. 112 113 Args: 114 path: Filepath to a folder where the data is downloaded for further processing. 115 split: The choice of data split. 116 download: Whether to download the data if it is not present. 117 118 Returns: 119 List of filepaths for the volumetric data. 120 """ 121 data_dir = get_longciu_data(path, download) 122 123 if split not in ("train", "val", "test"): 124 raise ValueError(f"'{split}' is not a valid split.") 125 126 volume_paths = natsorted(glob(os.path.join(data_dir, f"longciu_{split}.h5"))) 127 return volume_paths 128 129 130def get_longciu_dataset( 131 path: Union[os.PathLike, str], 132 patch_shape: Tuple[int, ...], 133 split: Literal["train", "val", "test"], 134 annotator: Literal["staple", "1", "2", "3"] = "staple", 135 resize_inputs: bool = False, 136 download: bool = False, 137 **kwargs 138) -> Dataset: 139 """Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT. 140 141 Args: 142 path: Filepath to a folder where the data is downloaded for further processing. 143 patch_shape: The patch shape to use for training. 144 split: The choice of data split. 145 annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation. 146 resize_inputs: Whether to resize the inputs to the patch shape. 147 download: Whether to download the data if it is not present. 148 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 149 150 Returns: 151 The segmentation dataset. 152 """ 153 volume_paths = get_longciu_paths(path, split, download) 154 155 if annotator not in ("staple", "1", "2", "3"): 156 raise ValueError(f"'{annotator}' is not a valid choice of annotator.") 157 158 if resize_inputs: 159 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 160 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 161 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 162 ) 163 164 return torch_em.default_segmentation_dataset( 165 raw_paths=volume_paths, 166 raw_key="raw", 167 label_paths=volume_paths, 168 label_key=f"labels/{annotator}", 169 patch_shape=patch_shape, 170 is_seg_dataset=True, 171 **kwargs, 172 ) 173 174 175def get_longciu_loader( 176 path: Union[os.PathLike, str], 177 batch_size: int, 178 patch_shape: Tuple[int, ...], 179 split: Literal["train", "val", "test"], 180 annotator: Literal["staple", "1", "2", "3"] = "staple", 181 resize_inputs: bool = False, 182 download: bool = False, 183 **kwargs 184) -> DataLoader: 185 """Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT. 186 187 Args: 188 path: Filepath to a folder where the data is downloaded for further processing. 189 batch_size: The batch size for training. 190 patch_shape: The patch shape to use for training. 191 split: The choice of data split. 192 annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation. 193 resize_inputs: Whether to resize the inputs to the patch shape. 194 download: Whether to download the data if it is not present. 195 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 196 197 Returns: 198 The DataLoader. 199 """ 200 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 201 dataset = get_longciu_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs) 202 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
84def get_longciu_data(path: Union[os.PathLike, str], download: bool = False) -> str: 85 """Download the LongCIU dataset. 86 87 Args: 88 path: Filepath to a folder where the data is downloaded for further processing. 89 download: Whether to download the data if it is not present. 90 91 Returns: 92 Filepath where the preprocessed data is stored. 93 """ 94 data_dir = os.path.join(path, "data") 95 if os.path.exists(data_dir): 96 return data_dir 97 98 os.makedirs(path, exist_ok=True) 99 100 for name, pid in FILE_PIDS.items(): 101 fpath = os.path.join(path, FNAMES[name]) 102 util.download_source(path=fpath, url=BASE_URL.format(pid), download=download, checksum=CHECKSUMS[name]) 103 104 _preprocess_data(path) 105 106 return data_dir
Download the LongCIU dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
109def get_longciu_paths( 110 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False 111) -> List[str]: 112 """Get paths to the LongCIU data. 113 114 Args: 115 path: Filepath to a folder where the data is downloaded for further processing. 116 split: The choice of data split. 117 download: Whether to download the data if it is not present. 118 119 Returns: 120 List of filepaths for the volumetric data. 121 """ 122 data_dir = get_longciu_data(path, download) 123 124 if split not in ("train", "val", "test"): 125 raise ValueError(f"'{split}' is not a valid split.") 126 127 volume_paths = natsorted(glob(os.path.join(data_dir, f"longciu_{split}.h5"))) 128 return volume_paths
Get paths to the LongCIU data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the volumetric data.
131def get_longciu_dataset( 132 path: Union[os.PathLike, str], 133 patch_shape: Tuple[int, ...], 134 split: Literal["train", "val", "test"], 135 annotator: Literal["staple", "1", "2", "3"] = "staple", 136 resize_inputs: bool = False, 137 download: bool = False, 138 **kwargs 139) -> Dataset: 140 """Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT. 141 142 Args: 143 path: Filepath to a folder where the data is downloaded for further processing. 144 patch_shape: The patch shape to use for training. 145 split: The choice of data split. 146 annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation. 147 resize_inputs: Whether to resize the inputs to the patch shape. 148 download: Whether to download the data if it is not present. 149 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 150 151 Returns: 152 The segmentation dataset. 153 """ 154 volume_paths = get_longciu_paths(path, split, download) 155 156 if annotator not in ("staple", "1", "2", "3"): 157 raise ValueError(f"'{annotator}' is not a valid choice of annotator.") 158 159 if resize_inputs: 160 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 161 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 162 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 163 ) 164 165 return torch_em.default_segmentation_dataset( 166 raw_paths=volume_paths, 167 raw_key="raw", 168 label_paths=volume_paths, 169 label_key=f"labels/{annotator}", 170 patch_shape=patch_shape, 171 is_seg_dataset=True, 172 **kwargs, 173 )
Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
176def get_longciu_loader( 177 path: Union[os.PathLike, str], 178 batch_size: int, 179 patch_shape: Tuple[int, ...], 180 split: Literal["train", "val", "test"], 181 annotator: Literal["staple", "1", "2", "3"] = "staple", 182 resize_inputs: bool = False, 183 download: bool = False, 184 **kwargs 185) -> DataLoader: 186 """Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT. 187 188 Args: 189 path: Filepath to a folder where the data is downloaded for further processing. 190 batch_size: The batch size for training. 191 patch_shape: The patch shape to use for training. 192 split: The choice of data split. 193 annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation. 194 resize_inputs: Whether to resize the inputs to the patch shape. 195 download: Whether to download the data if it is not present. 196 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 197 198 Returns: 199 The DataLoader. 200 """ 201 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 202 dataset = get_longciu_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs) 203 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.