torch_em.data.datasets.medical.longciu

The LongCIU dataset contains annotations for ground glass opacity and consolidation in chest CT scans of Long COVID patients.

This dataset is located at https://doi.org/10.25820/data.007301, and is released under the Open Data Commons Attribution License (ODC-By) v1.0. The dataset is from the publication https://doi.org/10.1038/s41597-025-04709-2. Please cite it if you use this dataset for your research.

  1"""The LongCIU dataset contains annotations for ground glass opacity and consolidation
  2in chest CT scans of Long COVID patients.
  3
  4This dataset is located at https://doi.org/10.25820/data.007301, and is released
  5under the Open Data Commons Attribution License (ODC-By) v1.0.
  6The dataset is from the publication https://doi.org/10.1038/s41597-025-04709-2.
  7Please cite it if you use this dataset for your research.
  8"""
  9
 10import os
 11import json
 12from glob import glob
 13from natsort import natsorted
 14from typing import Tuple, Union, Literal, List
 15
 16import numpy as np
 17
 18from torch.utils.data import Dataset, DataLoader
 19
 20import torch_em
 21
 22from .. import util
 23
 24
 25BASE_URL = "https://iro.uiowa.edu/view/fileRedirect?instCode=01IOWA_INST&download=true&filePid={}"
 26
 27FILE_PIDS = {
 28    "img": "13931395260002771",
 29    "staple": "13931395270002771",
 30    "1": "13931395300002771",
 31    "2": "13931395290002771",
 32    "3": "13931395280002771",
 33    "splits": "13931395240002771",
 34}
 35
 36FNAMES = {
 37    "img": "longciu_img.nii.gz",
 38    "staple": "longciu_STAPLE_tgt.nii.gz",
 39    "1": "longciu_1_tgt.nii.gz",
 40    "2": "longciu_2_tgt.nii.gz",
 41    "3": "longciu_3_tgt.nii.gz",
 42    "splits": "longciu_splits.json",
 43}
 44
 45CHECKSUMS = {
 46    "img": "af54aff713e7054d204b5e5e2aa0598c856006229901eea791ece8ae2efe8ab8",
 47    "staple": "6f0d64315c5010b296285f81a54adec0c6eae5fc643c4e21f4546129d743ea13",
 48    "1": "f9cd0e605f6fcd701a64a262ba15d43a46fe2f1b62a4045e83195e4962f022ed",
 49    "2": "4e54ec5add75b2ee48f34be6bd25c74f23d550a96430ff10c04e711383f9647f",
 50    "3": "aee04feddb18cbbe717418ce8b6838b58fc415f0b5b13b97b1c682faed1bb7cd",
 51    "splits": "7f4d62f33f68a16cd2df10cc320e087d43802970de63e6e459885f5c7e5604bf",
 52}
 53
 54
 55def _preprocess_data(path):
 56    import h5py
 57    import nibabel as nib
 58
 59    def _load(name):
 60        vol = nib.load(os.path.join(path, FNAMES[name])).get_fdata()
 61        return vol.transpose(2, 0, 1)
 62
 63    raw = _load("img").astype("float32")
 64    labels = {rater: np.rint(_load(rater)).astype("uint8") for rater in ["staple", "1", "2", "3"]}
 65
 66    with open(os.path.join(path, FNAMES["splits"])) as f:
 67        splits = json.load(f)
 68
 69    data_dir = os.path.join(path, "data")
 70    os.makedirs(data_dir, exist_ok=True)
 71
 72    for split, ids in splits.items():
 73        ids = sorted(ids)
 74        with h5py.File(os.path.join(data_dir, f"longciu_{split}.h5"), "w") as f:
 75            f.create_dataset("raw", data=raw[ids], compression="gzip")
 76            for rater, vol in labels.items():
 77                f.create_dataset(f"labels/{rater}", data=vol[ids], compression="gzip")
 78
 79    for name in ["img", "staple", "1", "2", "3"]:
 80        os.remove(os.path.join(path, FNAMES[name]))
 81
 82
 83def get_longciu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 84    """Download the LongCIU dataset.
 85
 86    Args:
 87        path: Filepath to a folder where the data is downloaded for further processing.
 88        download: Whether to download the data if it is not present.
 89
 90    Returns:
 91        Filepath where the preprocessed data is stored.
 92    """
 93    data_dir = os.path.join(path, "data")
 94    if os.path.exists(data_dir):
 95        return data_dir
 96
 97    os.makedirs(path, exist_ok=True)
 98
 99    for name, pid in FILE_PIDS.items():
100        fpath = os.path.join(path, FNAMES[name])
101        util.download_source(path=fpath, url=BASE_URL.format(pid), download=download, checksum=CHECKSUMS[name])
102
103    _preprocess_data(path)
104
105    return data_dir
106
107
108def get_longciu_paths(
109    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
110) -> List[str]:
111    """Get paths to the LongCIU data.
112
113    Args:
114        path: Filepath to a folder where the data is downloaded for further processing.
115        split: The choice of data split.
116        download: Whether to download the data if it is not present.
117
118    Returns:
119        List of filepaths for the volumetric data.
120    """
121    data_dir = get_longciu_data(path, download)
122
123    if split not in ("train", "val", "test"):
124        raise ValueError(f"'{split}' is not a valid split.")
125
126    volume_paths = natsorted(glob(os.path.join(data_dir, f"longciu_{split}.h5")))
127    return volume_paths
128
129
130def get_longciu_dataset(
131    path: Union[os.PathLike, str],
132    patch_shape: Tuple[int, ...],
133    split: Literal["train", "val", "test"],
134    annotator: Literal["staple", "1", "2", "3"] = "staple",
135    resize_inputs: bool = False,
136    download: bool = False,
137    **kwargs
138) -> Dataset:
139    """Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT.
140
141    Args:
142        path: Filepath to a folder where the data is downloaded for further processing.
143        patch_shape: The patch shape to use for training.
144        split: The choice of data split.
145        annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
146        resize_inputs: Whether to resize the inputs to the patch shape.
147        download: Whether to download the data if it is not present.
148        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
149
150    Returns:
151        The segmentation dataset.
152    """
153    volume_paths = get_longciu_paths(path, split, download)
154
155    if annotator not in ("staple", "1", "2", "3"):
156        raise ValueError(f"'{annotator}' is not a valid choice of annotator.")
157
158    if resize_inputs:
159        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
160        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
161            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
162        )
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=volume_paths,
166        raw_key="raw",
167        label_paths=volume_paths,
168        label_key=f"labels/{annotator}",
169        patch_shape=patch_shape,
170        is_seg_dataset=True,
171        **kwargs,
172    )
173
174
175def get_longciu_loader(
176    path: Union[os.PathLike, str],
177    batch_size: int,
178    patch_shape: Tuple[int, ...],
179    split: Literal["train", "val", "test"],
180    annotator: Literal["staple", "1", "2", "3"] = "staple",
181    resize_inputs: bool = False,
182    download: bool = False,
183    **kwargs
184) -> DataLoader:
185    """Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT.
186
187    Args:
188        path: Filepath to a folder where the data is downloaded for further processing.
189        batch_size: The batch size for training.
190        patch_shape: The patch shape to use for training.
191        split: The choice of data split.
192        annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
193        resize_inputs: Whether to resize the inputs to the patch shape.
194        download: Whether to download the data if it is not present.
195        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
196
197    Returns:
198        The DataLoader.
199    """
200    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
201    dataset = get_longciu_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs)
202    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
BASE_URL = 'https://iro.uiowa.edu/view/fileRedirect?instCode=01IOWA_INST&download=true&filePid={}'
FILE_PIDS = {'img': '13931395260002771', 'staple': '13931395270002771', '1': '13931395300002771', '2': '13931395290002771', '3': '13931395280002771', 'splits': '13931395240002771'}
FNAMES = {'img': 'longciu_img.nii.gz', 'staple': 'longciu_STAPLE_tgt.nii.gz', '1': 'longciu_1_tgt.nii.gz', '2': 'longciu_2_tgt.nii.gz', '3': 'longciu_3_tgt.nii.gz', 'splits': 'longciu_splits.json'}
CHECKSUMS = {'img': 'af54aff713e7054d204b5e5e2aa0598c856006229901eea791ece8ae2efe8ab8', 'staple': '6f0d64315c5010b296285f81a54adec0c6eae5fc643c4e21f4546129d743ea13', '1': 'f9cd0e605f6fcd701a64a262ba15d43a46fe2f1b62a4045e83195e4962f022ed', '2': '4e54ec5add75b2ee48f34be6bd25c74f23d550a96430ff10c04e711383f9647f', '3': 'aee04feddb18cbbe717418ce8b6838b58fc415f0b5b13b97b1c682faed1bb7cd', 'splits': '7f4d62f33f68a16cd2df10cc320e087d43802970de63e6e459885f5c7e5604bf'}
def get_longciu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 84def get_longciu_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 85    """Download the LongCIU dataset.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        Filepath where the preprocessed data is stored.
 93    """
 94    data_dir = os.path.join(path, "data")
 95    if os.path.exists(data_dir):
 96        return data_dir
 97
 98    os.makedirs(path, exist_ok=True)
 99
100    for name, pid in FILE_PIDS.items():
101        fpath = os.path.join(path, FNAMES[name])
102        util.download_source(path=fpath, url=BASE_URL.format(pid), download=download, checksum=CHECKSUMS[name])
103
104    _preprocess_data(path)
105
106    return data_dir

Download the LongCIU dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_longciu_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> List[str]:
109def get_longciu_paths(
110    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
111) -> List[str]:
112    """Get paths to the LongCIU data.
113
114    Args:
115        path: Filepath to a folder where the data is downloaded for further processing.
116        split: The choice of data split.
117        download: Whether to download the data if it is not present.
118
119    Returns:
120        List of filepaths for the volumetric data.
121    """
122    data_dir = get_longciu_data(path, download)
123
124    if split not in ("train", "val", "test"):
125        raise ValueError(f"'{split}' is not a valid split.")
126
127    volume_paths = natsorted(glob(os.path.join(data_dir, f"longciu_{split}.h5")))
128    return volume_paths

Get paths to the LongCIU data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the volumetric data.

def get_longciu_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], split: Literal['train', 'val', 'test'], annotator: Literal['staple', '1', '2', '3'] = 'staple', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
131def get_longciu_dataset(
132    path: Union[os.PathLike, str],
133    patch_shape: Tuple[int, ...],
134    split: Literal["train", "val", "test"],
135    annotator: Literal["staple", "1", "2", "3"] = "staple",
136    resize_inputs: bool = False,
137    download: bool = False,
138    **kwargs
139) -> Dataset:
140    """Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT.
141
142    Args:
143        path: Filepath to a folder where the data is downloaded for further processing.
144        patch_shape: The patch shape to use for training.
145        split: The choice of data split.
146        annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
147        resize_inputs: Whether to resize the inputs to the patch shape.
148        download: Whether to download the data if it is not present.
149        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
150
151    Returns:
152        The segmentation dataset.
153    """
154    volume_paths = get_longciu_paths(path, split, download)
155
156    if annotator not in ("staple", "1", "2", "3"):
157        raise ValueError(f"'{annotator}' is not a valid choice of annotator.")
158
159    if resize_inputs:
160        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
161        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
162            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
163        )
164
165    return torch_em.default_segmentation_dataset(
166        raw_paths=volume_paths,
167        raw_key="raw",
168        label_paths=volume_paths,
169        label_key=f"labels/{annotator}",
170        patch_shape=patch_shape,
171        is_seg_dataset=True,
172        **kwargs,
173    )

Get the LongCIU dataset for ground glass opacity and consolidation segmentation in chest CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_longciu_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], split: Literal['train', 'val', 'test'], annotator: Literal['staple', '1', '2', '3'] = 'staple', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
176def get_longciu_loader(
177    path: Union[os.PathLike, str],
178    batch_size: int,
179    patch_shape: Tuple[int, ...],
180    split: Literal["train", "val", "test"],
181    annotator: Literal["staple", "1", "2", "3"] = "staple",
182    resize_inputs: bool = False,
183    download: bool = False,
184    **kwargs
185) -> DataLoader:
186    """Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT.
187
188    Args:
189        path: Filepath to a folder where the data is downloaded for further processing.
190        batch_size: The batch size for training.
191        patch_shape: The patch shape to use for training.
192        split: The choice of data split.
193        annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
194        resize_inputs: Whether to resize the inputs to the patch shape.
195        download: Whether to download the data if it is not present.
196        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
197
198    Returns:
199        The DataLoader.
200    """
201    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
202    dataset = get_longciu_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs)
203    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the LongCIU dataloader for ground glass opacity and consolidation segmentation in chest CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • annotator: The choice of annotator providing the labels. Use 'staple' for the consensus segmentation.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.