torch_em.data.datasets.medical.denpar
The DenPAR dataset contains annotations for tooth segmentation in intraoral periapical (IOPA) radiographs.
The dataset is from the publication https://doi.org/10.1038/s41597-025-05906-9. Please cite it if you use this dataset in your research.
This module uses version 3 of the dataset (Zenodo record 16645076), which is openly downloadable. Earlier versions (v1: 14181645, v2: 13998619) are restricted and require a Zenodo access request.
The dataset also provides bone-level annotations and keypoint (CEJ, APEX) annotations, which are not exposed by this module; only the radiograph-wise (semantic) and tooth-wise (instance) tooth segmentation masks are used here.
1"""The DenPAR dataset contains annotations for tooth segmentation in intraoral periapical (IOPA) 2radiographs. 3 4The dataset is from the publication https://doi.org/10.1038/s41597-025-05906-9. Please cite it 5if you use this dataset in your research. 6 7This module uses version 3 of the dataset (Zenodo record 16645076), which is openly downloadable. 8Earlier versions (v1: 14181645, v2: 13998619) are restricted and require a Zenodo access request. 9 10The dataset also provides bone-level annotations and keypoint (CEJ, APEX) annotations, which are 11not exposed by this module; only the radiograph-wise (semantic) and tooth-wise (instance) tooth 12segmentation masks are used here. 13""" 14 15import os 16from glob import glob 17from pathlib import Path 18from natsort import natsorted 19from typing import Union, Tuple, Literal, List 20 21import numpy as np 22import imageio.v3 as imageio 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31URL = "https://zenodo.org/records/16645076/files/DenPAR%20Radiographs%20Dataset.zip" 32CHECKSUM = "b9edb55020f2cb971ba771b4cf5e4b65c4abb4df957310bd2eccc83d5a08b072" 33 34SPLITS = {"train": "Training", "val": "Validation", "test": "Testing"} 35 36 37def get_denpar_data(path: Union[os.PathLike, str], download: bool = False) -> str: 38 """Download the DenPAR dataset. 39 40 Args: 41 path: Filepath to a folder where the data is downloaded for further processing. 42 download: Whether to download the data if it is not present. 43 44 Returns: 45 Filepath where the data is downloaded. 46 """ 47 data_dir = os.path.join(path, "Dataset") 48 if os.path.exists(data_dir): 49 return data_dir 50 51 os.makedirs(path, exist_ok=True) 52 53 zip_path = os.path.join(path, "denpar_radiographs_dataset.zip") 54 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 55 util.unzip(zip_path=zip_path, dst=path) 56 57 return data_dir 58 59 60def _rasterize_instances(image_path, tooth_mask_dir, preprocessed_path): 61 image_id = Path(image_path).stem 62 tooth_mask_paths = natsorted(glob(os.path.join(tooth_mask_dir, image_id, "*.png"))) 63 64 shape = imageio.imread(image_path).shape[:2] 65 instances = np.zeros(shape, dtype="uint16") 66 for i, tooth_mask_path in enumerate(tooth_mask_paths, start=1): 67 mask = imageio.imread(tooth_mask_path) > 0 68 instances[mask] = i 69 70 imageio.imwrite(preprocessed_path, instances) 71 72 73def get_denpar_paths( 74 path: Union[os.PathLike, str], 75 split: Literal["train", "val", "test"], 76 label_choice: Literal["semantic", "instance"] = "semantic", 77 download: bool = False, 78) -> Tuple[List[str], List[str]]: 79 """Get paths to the DenPAR data. 80 81 Args: 82 path: Filepath to a folder where the data is downloaded for further processing. 83 split: The data split to use. Either 'train', 'val' or 'test'. 84 label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask, 85 one mask per radiograph) or 'instance' (individual tooth instances, rasterized from 86 the per-tooth masks into a single label map). 87 download: Whether to download the data if it is not present. 88 89 Returns: 90 List of filepaths for the image data. 91 List of filepaths for the label data. 92 """ 93 if split not in SPLITS: 94 raise ValueError(f"'{split}' is not a valid split. Please choose from {list(SPLITS.keys())}.") 95 96 if label_choice not in ("semantic", "instance"): 97 raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'semantic' or 'instance'.") 98 99 data_dir = get_denpar_data(path, download) 100 split_dir = os.path.join(data_dir, SPLITS[split]) 101 102 image_paths = natsorted(glob(os.path.join(split_dir, "Images", "*.jpg"))) 103 104 if label_choice == "semantic": 105 gt_dir = os.path.join(split_dir, "Masks (Radiograph-wise)") 106 gt_paths = [os.path.join(gt_dir, f"{Path(p).stem}.png") for p in image_paths] 107 image_paths = [p for p, g in zip(image_paths, gt_paths) if os.path.exists(g)] 108 gt_paths = [g for g in gt_paths if os.path.exists(g)] 109 110 else: 111 tooth_mask_dir = os.path.join(split_dir, "Masks (Tooth-wise)") 112 preprocessed_dir = os.path.join(split_dir, "preprocessed_instances") 113 os.makedirs(preprocessed_dir, exist_ok=True) 114 115 fimage_paths, gt_paths = [], [] 116 for image_path in image_paths: 117 image_id = Path(image_path).stem 118 if not os.path.exists(os.path.join(tooth_mask_dir, image_id)): 119 continue 120 121 gt_path = os.path.join(preprocessed_dir, f"{image_id}.tif") 122 if not os.path.exists(gt_path): 123 _rasterize_instances(image_path, tooth_mask_dir, gt_path) 124 125 fimage_paths.append(image_path) 126 gt_paths.append(gt_path) 127 128 image_paths = fimage_paths 129 130 return image_paths, gt_paths 131 132 133def get_denpar_dataset( 134 path: Union[os.PathLike, str], 135 patch_shape: Tuple[int, int], 136 split: Literal["train", "val", "test"], 137 label_choice: Literal["semantic", "instance"] = "semantic", 138 resize_inputs: bool = False, 139 download: bool = False, 140 **kwargs 141) -> Dataset: 142 """Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs. 143 144 Args: 145 path: Filepath to a folder where the data is downloaded for further processing. 146 patch_shape: The patch shape to use for training. 147 split: The data split to use. Either 'train', 'val' or 'test'. 148 label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'. 149 resize_inputs: Whether to resize the inputs to the patch shape. 150 download: Whether to download the data if it is not present. 151 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 152 153 Returns: 154 The segmentation dataset. 155 """ 156 image_paths, gt_paths = get_denpar_paths(path, split, label_choice, download) 157 158 if resize_inputs: 159 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 160 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 161 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 162 ) 163 164 return torch_em.default_segmentation_dataset( 165 raw_paths=image_paths, 166 raw_key=None, 167 label_paths=gt_paths, 168 label_key=None, 169 is_seg_dataset=False, 170 patch_shape=patch_shape, 171 **kwargs 172 ) 173 174 175def get_denpar_loader( 176 path: Union[os.PathLike, str], 177 batch_size: int, 178 patch_shape: Tuple[int, int], 179 split: Literal["train", "val", "test"], 180 label_choice: Literal["semantic", "instance"] = "semantic", 181 resize_inputs: bool = False, 182 download: bool = False, 183 **kwargs 184) -> DataLoader: 185 """Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs. 186 187 Args: 188 path: Filepath to a folder where the data is downloaded for further processing. 189 batch_size: The batch size for training. 190 patch_shape: The patch shape to use for training. 191 split: The data split to use. Either 'train', 'val' or 'test'. 192 label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'. 193 resize_inputs: Whether to resize the inputs to the patch shape. 194 download: Whether to download the data if it is not present. 195 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 196 197 Returns: 198 The DataLoader. 199 """ 200 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 201 dataset = get_denpar_dataset(path, patch_shape, split, label_choice, resize_inputs, download, **ds_kwargs) 202 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
38def get_denpar_data(path: Union[os.PathLike, str], download: bool = False) -> str: 39 """Download the DenPAR dataset. 40 41 Args: 42 path: Filepath to a folder where the data is downloaded for further processing. 43 download: Whether to download the data if it is not present. 44 45 Returns: 46 Filepath where the data is downloaded. 47 """ 48 data_dir = os.path.join(path, "Dataset") 49 if os.path.exists(data_dir): 50 return data_dir 51 52 os.makedirs(path, exist_ok=True) 53 54 zip_path = os.path.join(path, "denpar_radiographs_dataset.zip") 55 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 56 util.unzip(zip_path=zip_path, dst=path) 57 58 return data_dir
Download the DenPAR dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
74def get_denpar_paths( 75 path: Union[os.PathLike, str], 76 split: Literal["train", "val", "test"], 77 label_choice: Literal["semantic", "instance"] = "semantic", 78 download: bool = False, 79) -> Tuple[List[str], List[str]]: 80 """Get paths to the DenPAR data. 81 82 Args: 83 path: Filepath to a folder where the data is downloaded for further processing. 84 split: The data split to use. Either 'train', 'val' or 'test'. 85 label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask, 86 one mask per radiograph) or 'instance' (individual tooth instances, rasterized from 87 the per-tooth masks into a single label map). 88 download: Whether to download the data if it is not present. 89 90 Returns: 91 List of filepaths for the image data. 92 List of filepaths for the label data. 93 """ 94 if split not in SPLITS: 95 raise ValueError(f"'{split}' is not a valid split. Please choose from {list(SPLITS.keys())}.") 96 97 if label_choice not in ("semantic", "instance"): 98 raise ValueError(f"'{label_choice}' is not a valid label choice. Please choose 'semantic' or 'instance'.") 99 100 data_dir = get_denpar_data(path, download) 101 split_dir = os.path.join(data_dir, SPLITS[split]) 102 103 image_paths = natsorted(glob(os.path.join(split_dir, "Images", "*.jpg"))) 104 105 if label_choice == "semantic": 106 gt_dir = os.path.join(split_dir, "Masks (Radiograph-wise)") 107 gt_paths = [os.path.join(gt_dir, f"{Path(p).stem}.png") for p in image_paths] 108 image_paths = [p for p, g in zip(image_paths, gt_paths) if os.path.exists(g)] 109 gt_paths = [g for g in gt_paths if os.path.exists(g)] 110 111 else: 112 tooth_mask_dir = os.path.join(split_dir, "Masks (Tooth-wise)") 113 preprocessed_dir = os.path.join(split_dir, "preprocessed_instances") 114 os.makedirs(preprocessed_dir, exist_ok=True) 115 116 fimage_paths, gt_paths = [], [] 117 for image_path in image_paths: 118 image_id = Path(image_path).stem 119 if not os.path.exists(os.path.join(tooth_mask_dir, image_id)): 120 continue 121 122 gt_path = os.path.join(preprocessed_dir, f"{image_id}.tif") 123 if not os.path.exists(gt_path): 124 _rasterize_instances(image_path, tooth_mask_dir, gt_path) 125 126 fimage_paths.append(image_path) 127 gt_paths.append(gt_path) 128 129 image_paths = fimage_paths 130 131 return image_paths, gt_paths
Get paths to the DenPAR data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The data split to use. Either 'train', 'val' or 'test'.
- label_choice: The choice of segmentation labels. Either 'semantic' (binary tooth mask, one mask per radiograph) or 'instance' (individual tooth instances, rasterized from the per-tooth masks into a single label map).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
134def get_denpar_dataset( 135 path: Union[os.PathLike, str], 136 patch_shape: Tuple[int, int], 137 split: Literal["train", "val", "test"], 138 label_choice: Literal["semantic", "instance"] = "semantic", 139 resize_inputs: bool = False, 140 download: bool = False, 141 **kwargs 142) -> Dataset: 143 """Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs. 144 145 Args: 146 path: Filepath to a folder where the data is downloaded for further processing. 147 patch_shape: The patch shape to use for training. 148 split: The data split to use. Either 'train', 'val' or 'test'. 149 label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'. 150 resize_inputs: Whether to resize the inputs to the patch shape. 151 download: Whether to download the data if it is not present. 152 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 153 154 Returns: 155 The segmentation dataset. 156 """ 157 image_paths, gt_paths = get_denpar_paths(path, split, label_choice, download) 158 159 if resize_inputs: 160 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 161 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 162 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 163 ) 164 165 return torch_em.default_segmentation_dataset( 166 raw_paths=image_paths, 167 raw_key=None, 168 label_paths=gt_paths, 169 label_key=None, 170 is_seg_dataset=False, 171 patch_shape=patch_shape, 172 **kwargs 173 )
Get the DenPAR dataset for tooth segmentation in intraoral periapical radiographs.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The data split to use. Either 'train', 'val' or 'test'.
- label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
176def get_denpar_loader( 177 path: Union[os.PathLike, str], 178 batch_size: int, 179 patch_shape: Tuple[int, int], 180 split: Literal["train", "val", "test"], 181 label_choice: Literal["semantic", "instance"] = "semantic", 182 resize_inputs: bool = False, 183 download: bool = False, 184 **kwargs 185) -> DataLoader: 186 """Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs. 187 188 Args: 189 path: Filepath to a folder where the data is downloaded for further processing. 190 batch_size: The batch size for training. 191 patch_shape: The patch shape to use for training. 192 split: The data split to use. Either 'train', 'val' or 'test'. 193 label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'. 194 resize_inputs: Whether to resize the inputs to the patch shape. 195 download: Whether to download the data if it is not present. 196 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 197 198 Returns: 199 The DataLoader. 200 """ 201 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 202 dataset = get_denpar_dataset(path, patch_shape, split, label_choice, resize_inputs, download, **ds_kwargs) 203 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the DenPAR dataloader for tooth segmentation in intraoral periapical radiographs.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The data split to use. Either 'train', 'val' or 'test'.
- label_choice: The choice of segmentation labels. Either 'semantic' or 'instance'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.