torch_em.data.datasets.medical.nlstseg

The NLSTseg dataset contains pixel-level annotations of lung cancer lesions in low-dose CT scans of the National Lung Screening Trial (NLST).

The dataset consists of 605 patients with 715 manually annotated lesions (662 tumors and 53 nodules). Each patient is stored as '_CT.nii.gz' and '_tumor.nii.gz', the patients are distributed over 6 zip archives ('2_LungTumor.zip' - '7_LungTumor.zip', about 34 GB in total). The labels are instance labels: 0 is the background and each annotated lesion of a patient has its own id, starting from 1. Whether a lesion is a tumor or a nodule is only recorded in the metadata table '1_Table.zip' ('Label.xlsx', column 'labels_type'), not in the masks.

The data is located at https://doi.org/10.5281/zenodo.14838349, released under a CC-BY-4.0 license. NOTE: Older releases of this dataset are access restricted, this module uses the open release.

This dataset is from the publication https://doi.org/10.1038/s41597-025-05742-x. Please cite it if you use this dataset for your research.

  1"""The NLSTseg dataset contains pixel-level annotations of lung cancer lesions in low-dose CT scans
  2of the National Lung Screening Trial (NLST).
  3
  4The dataset consists of 605 patients with 715 manually annotated lesions (662 tumors and 53 nodules).
  5Each patient is stored as '<id>_CT.nii.gz' and '<id>_tumor.nii.gz', the patients are distributed over 6 zip archives
  6('2_LungTumor.zip' - '7_LungTumor.zip', about 34 GB in total). The labels are instance labels: 0 is the background
  7and each annotated lesion of a patient has its own id, starting from 1. Whether a lesion is a tumor or a nodule
  8is only recorded in the metadata table '1_Table.zip' ('Label.xlsx', column 'labels_type'), not in the masks.
  9
 10The data is located at https://doi.org/10.5281/zenodo.14838349, released under a CC-BY-4.0 license.
 11NOTE: Older releases of this dataset are access restricted, this module uses the open release.
 12
 13This dataset is from the publication https://doi.org/10.1038/s41597-025-05742-x.
 14Please cite it if you use this dataset for your research.
 15"""
 16
 17import os
 18from glob import glob
 19from natsort import natsorted
 20from typing import Union, Tuple, List, Optional, Sequence
 21
 22from torch.utils.data import Dataset, DataLoader
 23
 24import torch_em
 25
 26from .. import util
 27
 28
 29URL_BASE = "https://zenodo.org/records/14838349/files"
 30
 31CHECKSUMS = {
 32    2: None,
 33    3: "1ecb954e525e61b6f1a618a68dd955c79066c568bfad4d6e50c77887cdf618d5",
 34    4: None,
 35    5: None,
 36    6: None,
 37    7: None,
 38}
 39
 40
 41def get_nlstseg_data(
 42    path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False
 43) -> str:
 44    """Download the NLSTseg dataset.
 45
 46    NOTE: The full dataset is about 34 GB. Use `archives` to only download a subset of it.
 47
 48    Args:
 49        path: Filepath to a folder where the data is downloaded for further processing.
 50        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
 51            Each archive holds about 100 patients. By default all archives are used.
 52        download: Whether to download the data if it is not present.
 53
 54    Returns:
 55        Filepath where the data is downloaded.
 56    """
 57    archives = list(CHECKSUMS) if archives is None else list(archives)
 58    invalid = [n for n in archives if n not in CHECKSUMS]
 59    if invalid:
 60        raise ValueError(f"The archives {invalid} do not exist. Choose from {list(CHECKSUMS)}.")
 61
 62    data_dir = os.path.join(path, "data")
 63    for number in archives:
 64        if os.path.exists(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor")):
 65            continue
 66
 67        os.makedirs(data_dir, exist_ok=True)
 68        zip_path = os.path.join(path, f"{number}_LungTumor.zip")
 69        util.download_source(
 70            path=zip_path, url=f"{URL_BASE}/{number}_LungTumor.zip?download=1", download=download,
 71            checksum=CHECKSUMS[number],
 72        )
 73        util.unzip(zip_path=zip_path, dst=data_dir)
 74
 75    return data_dir
 76
 77
 78def get_nlstseg_paths(
 79    path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False
 80) -> Tuple[List[str], List[str]]:
 81    """Get paths to the NLSTseg data.
 82
 83    Args:
 84        path: Filepath to a folder where the data is downloaded for further processing.
 85        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
 86            By default all archives are used.
 87        download: Whether to download the data if it is not present.
 88
 89    Returns:
 90        List of filepaths for the image data.
 91        List of filepaths for the label data.
 92    """
 93    data_dir = get_nlstseg_data(path, archives, download)
 94
 95    numbers = list(CHECKSUMS) if archives is None else list(archives)
 96    raw_paths = []
 97    for number in numbers:
 98        raw_paths.extend(glob(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor", "*", "*_CT.nii.gz")))
 99    raw_paths = natsorted(raw_paths)
100    label_paths = [p.replace("_CT.nii.gz", "_tumor.nii.gz") for p in raw_paths]
101
102    assert len(raw_paths) > 0 and all(os.path.exists(p) for p in label_paths)
103
104    return raw_paths, label_paths
105
106
107def get_nlstseg_dataset(
108    path: Union[os.PathLike, str],
109    patch_shape: Tuple[int, int, int],
110    archives: Optional[Sequence[int]] = None,
111    resize_inputs: bool = False,
112    download: bool = False,
113    **kwargs
114) -> Dataset:
115    """Get the NLSTseg dataset for lung lesion segmentation in low-dose CT.
116
117    Args:
118        path: Filepath to a folder where the data is downloaded for further processing.
119        patch_shape: The patch shape to use for training.
120        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
121            By default all archives are used.
122        resize_inputs: Whether to resize the inputs to the patch shape.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    raw_paths, label_paths = get_nlstseg_paths(path, archives, download)
130
131    if resize_inputs:
132        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
133        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
134            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
135        )
136
137    return torch_em.default_segmentation_dataset(
138        raw_paths=raw_paths,
139        raw_key="data",
140        label_paths=label_paths,
141        label_key="data",
142        is_seg_dataset=True,
143        patch_shape=patch_shape,
144        ndim=3,
145        **kwargs
146    )
147
148
149def get_nlstseg_loader(
150    path: Union[os.PathLike, str],
151    batch_size: int,
152    patch_shape: Tuple[int, int, int],
153    archives: Optional[Sequence[int]] = None,
154    resize_inputs: bool = False,
155    download: bool = False,
156    **kwargs
157) -> DataLoader:
158    """Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        batch_size: The batch size for training.
163        patch_shape: The patch shape to use for training.
164        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
165            By default all archives are used.
166        resize_inputs: Whether to resize the inputs to the patch shape.
167        download: Whether to download the data if it is not present.
168        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
169
170    Returns:
171        The DataLoader.
172    """
173    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
174    dataset = get_nlstseg_dataset(path, patch_shape, archives, resize_inputs, download, **ds_kwargs)
175    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL_BASE = 'https://zenodo.org/records/14838349/files'
CHECKSUMS = {2: None, 3: '1ecb954e525e61b6f1a618a68dd955c79066c568bfad4d6e50c77887cdf618d5', 4: None, 5: None, 6: None, 7: None}
def get_nlstseg_data( path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False) -> str:
42def get_nlstseg_data(
43    path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False
44) -> str:
45    """Download the NLSTseg dataset.
46
47    NOTE: The full dataset is about 34 GB. Use `archives` to only download a subset of it.
48
49    Args:
50        path: Filepath to a folder where the data is downloaded for further processing.
51        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
52            Each archive holds about 100 patients. By default all archives are used.
53        download: Whether to download the data if it is not present.
54
55    Returns:
56        Filepath where the data is downloaded.
57    """
58    archives = list(CHECKSUMS) if archives is None else list(archives)
59    invalid = [n for n in archives if n not in CHECKSUMS]
60    if invalid:
61        raise ValueError(f"The archives {invalid} do not exist. Choose from {list(CHECKSUMS)}.")
62
63    data_dir = os.path.join(path, "data")
64    for number in archives:
65        if os.path.exists(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor")):
66            continue
67
68        os.makedirs(data_dir, exist_ok=True)
69        zip_path = os.path.join(path, f"{number}_LungTumor.zip")
70        util.download_source(
71            path=zip_path, url=f"{URL_BASE}/{number}_LungTumor.zip?download=1", download=download,
72            checksum=CHECKSUMS[number],
73        )
74        util.unzip(zip_path=zip_path, dst=data_dir)
75
76    return data_dir

Download the NLSTseg dataset.

NOTE: The full dataset is about 34 GB. Use archives to only download a subset of it.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • archives: The numbers of the archives ('_LungTumor.zip', from 2 to 7) to use. Each archive holds about 100 patients. By default all archives are used.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_nlstseg_paths( path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 79def get_nlstseg_paths(
 80    path: Union[os.PathLike, str], archives: Optional[Sequence[int]] = None, download: bool = False
 81) -> Tuple[List[str], List[str]]:
 82    """Get paths to the NLSTseg data.
 83
 84    Args:
 85        path: Filepath to a folder where the data is downloaded for further processing.
 86        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
 87            By default all archives are used.
 88        download: Whether to download the data if it is not present.
 89
 90    Returns:
 91        List of filepaths for the image data.
 92        List of filepaths for the label data.
 93    """
 94    data_dir = get_nlstseg_data(path, archives, download)
 95
 96    numbers = list(CHECKSUMS) if archives is None else list(archives)
 97    raw_paths = []
 98    for number in numbers:
 99        raw_paths.extend(glob(os.path.join(data_dir, f"NLSTseg_{number}_LungTumor", "*", "*_CT.nii.gz")))
100    raw_paths = natsorted(raw_paths)
101    label_paths = [p.replace("_CT.nii.gz", "_tumor.nii.gz") for p in raw_paths]
102
103    assert len(raw_paths) > 0 and all(os.path.exists(p) for p in label_paths)
104
105    return raw_paths, label_paths

Get paths to the NLSTseg data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • archives: The numbers of the archives ('_LungTumor.zip', from 2 to 7) to use. By default all archives are used.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_nlstseg_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], archives: Optional[Sequence[int]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
108def get_nlstseg_dataset(
109    path: Union[os.PathLike, str],
110    patch_shape: Tuple[int, int, int],
111    archives: Optional[Sequence[int]] = None,
112    resize_inputs: bool = False,
113    download: bool = False,
114    **kwargs
115) -> Dataset:
116    """Get the NLSTseg dataset for lung lesion segmentation in low-dose CT.
117
118    Args:
119        path: Filepath to a folder where the data is downloaded for further processing.
120        patch_shape: The patch shape to use for training.
121        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
122            By default all archives are used.
123        resize_inputs: Whether to resize the inputs to the patch shape.
124        download: Whether to download the data if it is not present.
125        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
126
127    Returns:
128        The segmentation dataset.
129    """
130    raw_paths, label_paths = get_nlstseg_paths(path, archives, download)
131
132    if resize_inputs:
133        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
134        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
135            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
136        )
137
138    return torch_em.default_segmentation_dataset(
139        raw_paths=raw_paths,
140        raw_key="data",
141        label_paths=label_paths,
142        label_key="data",
143        is_seg_dataset=True,
144        patch_shape=patch_shape,
145        ndim=3,
146        **kwargs
147    )

Get the NLSTseg dataset for lung lesion segmentation in low-dose CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • archives: The numbers of the archives ('_LungTumor.zip', from 2 to 7) to use. By default all archives are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_nlstseg_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], archives: Optional[Sequence[int]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
150def get_nlstseg_loader(
151    path: Union[os.PathLike, str],
152    batch_size: int,
153    patch_shape: Tuple[int, int, int],
154    archives: Optional[Sequence[int]] = None,
155    resize_inputs: bool = False,
156    download: bool = False,
157    **kwargs
158) -> DataLoader:
159    """Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT.
160
161    Args:
162        path: Filepath to a folder where the data is downloaded for further processing.
163        batch_size: The batch size for training.
164        patch_shape: The patch shape to use for training.
165        archives: The numbers of the archives ('<number>_LungTumor.zip', from 2 to 7) to use.
166            By default all archives are used.
167        resize_inputs: Whether to resize the inputs to the patch shape.
168        download: Whether to download the data if it is not present.
169        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
170
171    Returns:
172        The DataLoader.
173    """
174    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
175    dataset = get_nlstseg_dataset(path, patch_shape, archives, resize_inputs, download, **ds_kwargs)
176    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the NLSTseg dataloader for lung lesion segmentation in low-dose CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • archives: The numbers of the archives ('_LungTumor.zip', from 2 to 7) to use. By default all archives are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.