torch_em.data.datasets.medical.denpar

The DenPAR dataset contains annotations for tooth segmentation in intraoral periapical (IOPA) radiographs.

The dataset is from the publication https://doi.org/10.1038/s41597-025-05906-9. Please cite it if you use this dataset in your research.

This module uses version 3 of the dataset (Zenodo record 16645076), which is openly downloadable. Earlier versions (v1: 14181645, v2: 13998619) are restricted and require a Zenodo access request.

The dataset also provides bone-level annotations and keypoint (CEJ, APEX) annotations, which are not exposed by this module; only the radiograph-wise (semantic) and tooth-wise (instance) tooth segmentation masks are used here.

  1"""The DenPAR dataset contains annotations for tooth segmentation in intraoral periapical (IOPA)
  2radiographs.
  3
  4The dataset is from the publication https://doi.org/10.1038/s41597-025-05906-9. Please cite it
  5if you use this dataset in your research.
  6
  7This module uses version 3 of the dataset (Zenodo record 16645076), which is openly downloadable.
  8Earlier versions (v1: 14181645, v2: 13998619) are restricted and require a Zenodo access request.
  9
 10The dataset also provides bone-level annotations and keypoint (CEJ, APEX) annotations, which are
 11not exposed by this module; only the radiograph-wise (semantic) and tooth-wise (instance) tooth
 12segmentation masks are used here.
 13"""
 14
 15import os
 16from glob import glob
 17from pathlib import Path
 18from natsort import natsorted
 19from typing import Union, Tuple, Literal, List
 20
 21import numpy as np
 22import imageio.v3 as imageio
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URL = "https://zenodo.org/records/16645076/files/DenPAR%20Radiographs%20Dataset.zip"
 32CHECKSUM = "b9edb55020f2cb971ba771b4cf5e4b65c4abb4df957310bd2eccc83d5a08b072"
 33
 34SPLITS = {"train": "Training", "val": "Validation", "test": "Testing"}
 35
 36
 37def get_denpar_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 38    """Download the DenPAR dataset.
 39
 40    Args:
 41        path: Filepath to a folder where the data is downloaded for further processing.
 42        download: Whether to download the data if it is not present.
 43
 44    Returns:
 45        Filepath where the data is downloaded.
 46    """
 47    data_dir = os.path.join(path, "Dataset")
 48    if os.path.exists(data_dir):
 49        return data_dir
 50
 51    os.makedirs(path, exist_ok=True)
 52
 53    zip_path = os.path.join(path, "denpar_radiographs_dataset.zip")
 54    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 55    util.unzip(zip_path=zip_path, dst=path)
 56
 57    return data_dir
 58
 59
 60def _rasterize_instances(image_path, tooth_mask_dir, preprocessed_path):
 61    image_id = Path(image_path).stem
 62    tooth_mask_paths = natsorted(glob(os.path.join(tooth_mask_dir, image_id, "*.png")))
 63
 64    shape = imageio.imread(image_path).shape[:2]
 65    instances = np.zeros(shape, dtype="uint16")
 66    for i, tooth_mask_path in enumerate(tooth_mask_paths, start=1):
 67        mask = imageio.imread(tooth_mask_path) > 0
 68        instances[mask] = i
 69
 70    imageio.imwrite(preprocessed_path, instances)
 71
 72
 73def get_denpar_paths(
 74    path: Union[os.PathLike, str],
 75    split: Literal["train", "val", "test"],
 76    label_choice: Literal["semantic", "instance"] = "semantic",
 77    download: bool = False,
 78) -> Tuple[List[str], List[str]]:
 79    """Get paths to the DenPAR data.
 80
 81    Args:
 82        path: Filepath to a folder where the data is downloaded for further processing.
 83        split: The data split to use. Either 'train', 'val' or 'test'.
 84        label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask,
 85            one mask per radiograph) or 'instance' (individual tooth instances, rasterized from
 86            the per-tooth masks into a single label map).
 87        download: Whether to download the data if it is not present.
 88
 89    Returns:
 90        List of filepaths for the image data.
 91        List of filepaths for the label data.
 92    """
 93    if split not in SPLITS:
 94        raise ValueError(f"'{split}' is not a valid split. Please choose from {list(SPLITS.keys())}.")
 95
 96    if label_choice not in ("semantic", "instance"):
 97        raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'semantic' or 'instance'.")
 98
 99    data_dir = get_denpar_data(path, download)
100    split_dir = os.path.join(data_dir, SPLITS[split])
101
102    image_paths = natsorted(glob(os.path.join(split_dir, "Images", "*.jpg")))
103
104    if label_choice == "semantic":
105        gt_dir = os.path.join(split_dir, "Masks (Radiograph-wise)")
106        gt_paths = [os.path.join(gt_dir, f"{Path(p).stem}.png") for p in image_paths]
107        image_paths = [p for p, g in zip(image_paths, gt_paths) if os.path.exists(g)]
108        gt_paths = [g for g in gt_paths if os.path.exists(g)]
109
110    else:
111        tooth_mask_dir = os.path.join(split_dir, "Masks (Tooth-wise)")
112        preprocessed_dir = os.path.join(split_dir, "preprocessed_instances")
113        os.makedirs(preprocessed_dir, exist_ok=True)
114
115        fimage_paths, gt_paths = [], []
116        for image_path in image_paths:
117            image_id = Path(image_path).stem
118            if not os.path.exists(os.path.join(tooth_mask_dir, image_id)):
119                continue
120
121            gt_path = os.path.join(preprocessed_dir, f"{image_id}.tif")
122            if not os.path.exists(gt_path):
123                _rasterize_instances(image_path, tooth_mask_dir, gt_path)
124
125            fimage_paths.append(image_path)
126            gt_paths.append(gt_path)
127
128        image_paths = fimage_paths
129
130    return image_paths, gt_paths
131
132
133def get_denpar_dataset(
134    path: Union[os.PathLike, str],
135    patch_shape: Tuple[int, int],
136    split: Literal["train", "val", "test"],
137    label_choice: Literal["semantic", "instance"] = "semantic",
138    resize_inputs: bool = False,
139    download: bool = False,
140    **kwargs
141) -> Dataset:
142    """Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs.
143
144    Args:
145        path: Filepath to a folder where the data is downloaded for further processing.
146        patch_shape: The patch shape to use for training.
147        split: The data split to use. Either 'train', 'val' or 'test'.
148        label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
149        resize_inputs: Whether to resize the inputs to the patch shape.
150        download: Whether to download the data if it is not present.
151        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
152
153    Returns:
154        The segmentation dataset.
155    """
156    image_paths, gt_paths = get_denpar_paths(path, split, label_choice, download)
157
158    if resize_inputs:
159        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
160        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
161            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
162        )
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=image_paths,
166        raw_key=None,
167        label_paths=gt_paths,
168        label_key=None,
169        is_seg_dataset=False,
170        patch_shape=patch_shape,
171        **kwargs
172    )
173
174
175def get_denpar_loader(
176    path: Union[os.PathLike, str],
177    batch_size: int,
178    patch_shape: Tuple[int, int],
179    split: Literal["train", "val", "test"],
180    label_choice: Literal["semantic", "instance"] = "semantic",
181    resize_inputs: bool = False,
182    download: bool = False,
183    **kwargs
184) -> DataLoader:
185    """Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs.
186
187    Args:
188        path: Filepath to a folder where the data is downloaded for further processing.
189        batch_size: The batch size for training.
190        patch_shape: The patch shape to use for training.
191        split: The data split to use. Either 'train', 'val' or 'test'.
192        label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
193        resize_inputs: Whether to resize the inputs to the patch shape.
194        download: Whether to download the data if it is not present.
195        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
196
197    Returns:
198        The DataLoader.
199    """
200    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
201    dataset = get_denpar_dataset(path, patch_shape, split, label_choice, resize_inputs, download, **ds_kwargs)
202    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/16645076/files/DenPAR%20Radiographs%20Dataset.zip'
CHECKSUM = 'b9edb55020f2cb971ba771b4cf5e4b65c4abb4df957310bd2eccc83d5a08b072'
SPLITS = {'train': 'Training', 'val': 'Validation', 'test': 'Testing'}
def get_denpar_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38def get_denpar_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39    """Download the DenPAR dataset.
40
41    Args:
42        path: Filepath to a folder where the data is downloaded for further processing.
43        download: Whether to download the data if it is not present.
44
45    Returns:
46        Filepath where the data is downloaded.
47    """
48    data_dir = os.path.join(path, "Dataset")
49    if os.path.exists(data_dir):
50        return data_dir
51
52    os.makedirs(path, exist_ok=True)
53
54    zip_path = os.path.join(path, "denpar_radiographs_dataset.zip")
55    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
56    util.unzip(zip_path=zip_path, dst=path)
57
58    return data_dir

Download the DenPAR dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_denpar_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], label_choice: Literal['semantic', 'instance'] = 'semantic', download: bool = False) -> Tuple[List[str], List[str]]:
 74def get_denpar_paths(
 75    path: Union[os.PathLike, str],
 76    split: Literal["train", "val", "test"],
 77    label_choice: Literal["semantic", "instance"] = "semantic",
 78    download: bool = False,
 79) -> Tuple[List[str], List[str]]:
 80    """Get paths to the DenPAR data.
 81
 82    Args:
 83        path: Filepath to a folder where the data is downloaded for further processing.
 84        split: The data split to use. Either 'train', 'val' or 'test'.
 85        label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask,
 86            one mask per radiograph) or 'instance' (individual tooth instances, rasterized from
 87            the per-tooth masks into a single label map).
 88        download: Whether to download the data if it is not present.
 89
 90    Returns:
 91        List of filepaths for the image data.
 92        List of filepaths for the label data.
 93    """
 94    if split not in SPLITS:
 95        raise ValueError(f"'{split}' is not a valid split. Please choose from {list(SPLITS.keys())}.")
 96
 97    if label_choice not in ("semantic", "instance"):
 98        raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'semantic' or 'instance'.")
 99
100    data_dir = get_denpar_data(path, download)
101    split_dir = os.path.join(data_dir, SPLITS[split])
102
103    image_paths = natsorted(glob(os.path.join(split_dir, "Images", "*.jpg")))
104
105    if label_choice == "semantic":
106        gt_dir = os.path.join(split_dir, "Masks (Radiograph-wise)")
107        gt_paths = [os.path.join(gt_dir, f"{Path(p).stem}.png") for p in image_paths]
108        image_paths = [p for p, g in zip(image_paths, gt_paths) if os.path.exists(g)]
109        gt_paths = [g for g in gt_paths if os.path.exists(g)]
110
111    else:
112        tooth_mask_dir = os.path.join(split_dir, "Masks (Tooth-wise)")
113        preprocessed_dir = os.path.join(split_dir, "preprocessed_instances")
114        os.makedirs(preprocessed_dir, exist_ok=True)
115
116        fimage_paths, gt_paths = [], []
117        for image_path in image_paths:
118            image_id = Path(image_path).stem
119            if not os.path.exists(os.path.join(tooth_mask_dir, image_id)):
120                continue
121
122            gt_path = os.path.join(preprocessed_dir, f"{image_id}.tif")
123            if not os.path.exists(gt_path):
124                _rasterize_instances(image_path, tooth_mask_dir, gt_path)
125
126            fimage_paths.append(image_path)
127            gt_paths.append(gt_path)
128
129        image_paths = fimage_paths
130
131    return image_paths, gt_paths

Get paths to the DenPAR data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The data split to use. Either 'train', 'val' or 'test'.
  • label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask, one mask per radiograph) or 'instance' (individual tooth instances, rasterized from the per-tooth masks into a single label map).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_denpar_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], label_choice: Literal['semantic', 'instance'] = 'semantic', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
134def get_denpar_dataset(
135    path: Union[os.PathLike, str],
136    patch_shape: Tuple[int, int],
137    split: Literal["train", "val", "test"],
138    label_choice: Literal["semantic", "instance"] = "semantic",
139    resize_inputs: bool = False,
140    download: bool = False,
141    **kwargs
142) -> Dataset:
143    """Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs.
144
145    Args:
146        path: Filepath to a folder where the data is downloaded for further processing.
147        patch_shape: The patch shape to use for training.
148        split: The data split to use. Either 'train', 'val' or 'test'.
149        label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
150        resize_inputs: Whether to resize the inputs to the patch shape.
151        download: Whether to download the data if it is not present.
152        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
153
154    Returns:
155        The segmentation dataset.
156    """
157    image_paths, gt_paths = get_denpar_paths(path, split, label_choice, download)
158
159    if resize_inputs:
160        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
161        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
162            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
163        )
164
165    return torch_em.default_segmentation_dataset(
166        raw_paths=image_paths,
167        raw_key=None,
168        label_paths=gt_paths,
169        label_key=None,
170        is_seg_dataset=False,
171        patch_shape=patch_shape,
172        **kwargs
173    )

Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The data split to use. Either 'train', 'val' or 'test'.
  • label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_denpar_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], label_choice: Literal['semantic', 'instance'] = 'semantic', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
176def get_denpar_loader(
177    path: Union[os.PathLike, str],
178    batch_size: int,
179    patch_shape: Tuple[int, int],
180    split: Literal["train", "val", "test"],
181    label_choice: Literal["semantic", "instance"] = "semantic",
182    resize_inputs: bool = False,
183    download: bool = False,
184    **kwargs
185) -> DataLoader:
186    """Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs.
187
188    Args:
189        path: Filepath to a folder where the data is downloaded for further processing.
190        batch_size: The batch size for training.
191        patch_shape: The patch shape to use for training.
192        split: The data split to use. Either 'train', 'val' or 'test'.
193        label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
194        resize_inputs: Whether to resize the inputs to the patch shape.
195        download: Whether to download the data if it is not present.
196        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
197
198    Returns:
199        The DataLoader.
200    """
201    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
202    dataset = get_denpar_dataset(path, patch_shape, split, label_choice, resize_inputs, download, **ds_kwargs)
203    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The data split to use. Either 'train', 'val' or 'test'.
  • label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.