torch_em.data.datasets.medical.nlstseg
The NLSTseg dataset contains pixel-level annotations of lung cancer lesions in low-dose CT scans of the National Lung Screening Trial (NLST).
The dataset consists of 605 patients with 715 manually annotated lesions (662 tumors and 53 nodules).
Each patient is stored as '
The data is located at https://doi.org/10.5281/zenodo.14838349, released under a CC-BY-4.0 license. NOTE: Older releases of this dataset are access restricted, this module uses the open release.
This dataset is from the publication https://doi.org/10.1038/s41597-025-05742-x. Please cite it if you use this dataset for your research.
1"""The NLSTseg dataset contains pixel-level annotations of lung cancer lesions in low-dose CT scans 2of the National Lung Screening Trial (NLST). 3 4The dataset consists of 605 patients with 715 manually annotated lesions (662 tumors and 53 nodules). 5Each patient is stored as '<id>_CT.nii.gz' and '<id>_tumor.nii.gz', the patients are distributed over 6 zip archives 6('2_LungTumor.zip' - '7_LungTumor.zip', about 34 GB in total). The labels are instance labels: 0 is the background 7and each annotated lesion of a patient has its own id, starting from 1. Whether a lesion is a tumor or a nodule 8is only recorded in the metadata table '1_Table.zip' ('Label.xlsx', column 'labels_type'), not in the masks. 9 10The data is located at https://doi.org/10.5281/zenodo.14838349, released under a CC-BY-4.0 license. 11NOTE: Older releases of this dataset are access restricted, this module uses the open release. 12 13This dataset is from the publication https://doi.org/10.1038/s41597-025-05742-x. 14Please cite it if you use this dataset for your research. 15""" 16 17import os 18from glob import glob 19from natsort import natsorted 20from typing import Union, Tuple, List, Optional, Sequence 21 22from torch.utils.data import Dataset, DataLoader 23 24import torch_em 25 26from .. import util 27 28 29URL_BASE = "https://zenodo.org/records/14838349/files" 30 31CHECKSUMS = { 32 2: None, 33 3: "1ecb954e525e61b6f1a618a68dd955c79066c568bfad4d6e50c77887cdf618d5", 34 4: None, 35 5: None, 36 6: None, 37 7: None, 38} 39 40 41def get_nlstseg_data( 42 path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False 43) -> str: 44 """Download the NLSTseg dataset. 45 46 NOTE: The full dataset is about 34 GB. Use `archives` to only download a subset of it. 47 48 Args: 49 path: Filepath to a folder where the data is downloaded for further processing. 50 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 51 Each archive holds about 100 patients. By default all archives are used. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 archives = list(CHECKSUMS) if archives is None else list(archives) 58 invalid = [n for n in archives if n not in CHECKSUMS] 59 if invalid: 60 raise ValueError(f"The archives {invalid} do not exist. Choose from {list(CHECKSUMS)}.") 61 62 data_dir = os.path.join(path, "data") 63 for number in archives: 64 if os.path.exists(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor")): 65 continue 66 67 os.makedirs(data_dir, exist_ok=True) 68 zip_path = os.path.join(path, f"{number}_LungTumor.zip") 69 util.download_source( 70 path=zip_path, url=f"{URL_BASE}/{number}_LungTumor.zip?download=1", download=download, 71 checksum=CHECKSUMS[number], 72 ) 73 util.unzip(zip_path=zip_path, dst=data_dir) 74 75 return data_dir 76 77 78def get_nlstseg_paths( 79 path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False 80) -> Tuple[List[str], List[str]]: 81 """Get paths to the NLSTseg data. 82 83 Args: 84 path: Filepath to a folder where the data is downloaded for further processing. 85 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 86 By default all archives are used. 87 download: Whether to download the data if it is not present. 88 89 Returns: 90 List of filepaths for the image data. 91 List of filepaths for the label data. 92 """ 93 data_dir = get_nlstseg_data(path, archives, download) 94 95 numbers = list(CHECKSUMS) if archives is None else list(archives) 96 raw_paths = [] 97 for number in numbers: 98 raw_paths.extend(glob(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor", "*", "*_CT.nii.gz"))) 99 raw_paths = natsorted(raw_paths) 100 label_paths = [p.replace("_CT.nii.gz", "_tumor.nii.gz") for p in raw_paths] 101 102 assert len(raw_paths) > 0 and all(os.path.exists(p) for p in label_paths) 103 104 return raw_paths, label_paths 105 106 107def get_nlstseg_dataset( 108 path: Union[os.PathLike, str], 109 patch_shape: Tuple[int, int, int], 110 archives: Optional[Sequence[int]] = None, 111 resize_inputs: bool = False, 112 download: bool = False, 113 **kwargs 114) -> Dataset: 115 """Get the NLSTseg dataset for lung lesion segmentation in low-dose CT. 116 117 Args: 118 path: Filepath to a folder where the data is downloaded for further processing. 119 patch_shape: The patch shape to use for training. 120 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 121 By default all archives are used. 122 resize_inputs: Whether to resize the inputs to the patch shape. 123 download: Whether to download the data if it is not present. 124 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 125 126 Returns: 127 The segmentation dataset. 128 """ 129 raw_paths, label_paths = get_nlstseg_paths(path, archives, download) 130 131 if resize_inputs: 132 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 133 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 134 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 135 ) 136 137 return torch_em.default_segmentation_dataset( 138 raw_paths=raw_paths, 139 raw_key="data", 140 label_paths=label_paths, 141 label_key="data", 142 is_seg_dataset=True, 143 patch_shape=patch_shape, 144 ndim=3, 145 **kwargs 146 ) 147 148 149def get_nlstseg_loader( 150 path: Union[os.PathLike, str], 151 batch_size: int, 152 patch_shape: Tuple[int, int, int], 153 archives: Optional[Sequence[int]] = None, 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> DataLoader: 158 """Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 batch_size: The batch size for training. 163 patch_shape: The patch shape to use for training. 164 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 165 By default all archives are used. 166 resize_inputs: Whether to resize the inputs to the patch shape. 167 download: Whether to download the data if it is not present. 168 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 169 170 Returns: 171 The DataLoader. 172 """ 173 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 174 dataset = get_nlstseg_dataset(path, patch_shape, archives, resize_inputs, download, **ds_kwargs) 175 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
42def get_nlstseg_data( 43 path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False 44) -> str: 45 """Download the NLSTseg dataset. 46 47 NOTE: The full dataset is about 34 GB. Use `archives` to only download a subset of it. 48 49 Args: 50 path: Filepath to a folder where the data is downloaded for further processing. 51 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 52 Each archive holds about 100 patients. By default all archives are used. 53 download: Whether to download the data if it is not present. 54 55 Returns: 56 Filepath where the data is downloaded. 57 """ 58 archives = list(CHECKSUMS) if archives is None else list(archives) 59 invalid = [n for n in archives if n not in CHECKSUMS] 60 if invalid: 61 raise ValueError(f"The archives {invalid} do not exist. Choose from {list(CHECKSUMS)}.") 62 63 data_dir = os.path.join(path, "data") 64 for number in archives: 65 if os.path.exists(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor")): 66 continue 67 68 os.makedirs(data_dir, exist_ok=True) 69 zip_path = os.path.join(path, f"{number}_LungTumor.zip") 70 util.download_source( 71 path=zip_path, url=f"{URL_BASE}/{number}_LungTumor.zip?download=1", download=download, 72 checksum=CHECKSUMS[number], 73 ) 74 util.unzip(zip_path=zip_path, dst=data_dir) 75 76 return data_dir
Download the NLSTseg dataset.
NOTE: The full dataset is about 34 GB. Use archives to only download a subset of it.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- archives: The numbers of the archives ('
_LungTumor.zip', from 2 to 7) to use. Each archive holds about 100 patients. By default all archives are used. - download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
79def get_nlstseg_paths( 80 path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False 81) -> Tuple[List[str], List[str]]: 82 """Get paths to the NLSTseg data. 83 84 Args: 85 path: Filepath to a folder where the data is downloaded for further processing. 86 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 87 By default all archives are used. 88 download: Whether to download the data if it is not present. 89 90 Returns: 91 List of filepaths for the image data. 92 List of filepaths for the label data. 93 """ 94 data_dir = get_nlstseg_data(path, archives, download) 95 96 numbers = list(CHECKSUMS) if archives is None else list(archives) 97 raw_paths = [] 98 for number in numbers: 99 raw_paths.extend(glob(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor", "*", "*_CT.nii.gz"))) 100 raw_paths = natsorted(raw_paths) 101 label_paths = [p.replace("_CT.nii.gz", "_tumor.nii.gz") for p in raw_paths] 102 103 assert len(raw_paths) > 0 and all(os.path.exists(p) for p in label_paths) 104 105 return raw_paths, label_paths
Get paths to the NLSTseg data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- archives: The numbers of the archives ('
_LungTumor.zip', from 2 to 7) to use. By default all archives are used. - download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
108def get_nlstseg_dataset( 109 path: Union[os.PathLike, str], 110 patch_shape: Tuple[int, int, int], 111 archives: Optional[Sequence[int]] = None, 112 resize_inputs: bool = False, 113 download: bool = False, 114 **kwargs 115) -> Dataset: 116 """Get the NLSTseg dataset for lung lesion segmentation in low-dose CT. 117 118 Args: 119 path: Filepath to a folder where the data is downloaded for further processing. 120 patch_shape: The patch shape to use for training. 121 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 122 By default all archives are used. 123 resize_inputs: Whether to resize the inputs to the patch shape. 124 download: Whether to download the data if it is not present. 125 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 126 127 Returns: 128 The segmentation dataset. 129 """ 130 raw_paths, label_paths = get_nlstseg_paths(path, archives, download) 131 132 if resize_inputs: 133 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 134 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 135 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 136 ) 137 138 return torch_em.default_segmentation_dataset( 139 raw_paths=raw_paths, 140 raw_key="data", 141 label_paths=label_paths, 142 label_key="data", 143 is_seg_dataset=True, 144 patch_shape=patch_shape, 145 ndim=3, 146 **kwargs 147 )
Get the NLSTseg dataset for lung lesion segmentation in low-dose CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- archives: The numbers of the archives ('
_LungTumor.zip', from 2 to 7) to use. By default all archives are used. - resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
150def get_nlstseg_loader( 151 path: Union[os.PathLike, str], 152 batch_size: int, 153 patch_shape: Tuple[int, int, int], 154 archives: Optional[Sequence[int]] = None, 155 resize_inputs: bool = False, 156 download: bool = False, 157 **kwargs 158) -> DataLoader: 159 """Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT. 160 161 Args: 162 path: Filepath to a folder where the data is downloaded for further processing. 163 batch_size: The batch size for training. 164 patch_shape: The patch shape to use for training. 165 archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use. 166 By default all archives are used. 167 resize_inputs: Whether to resize the inputs to the patch shape. 168 download: Whether to download the data if it is not present. 169 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 170 171 Returns: 172 The DataLoader. 173 """ 174 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 175 dataset = get_nlstseg_dataset(path, patch_shape, archives, resize_inputs, download, **ds_kwargs) 176 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- archives: The numbers of the archives ('
_LungTumor.zip', from 2 to 7) to use. By default all archives are used. - resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.