torch_em.data.datasets.medical.pfus1
PFUS1 is a dataset for segmentation of pelvic floor anatomical structures in transperineal ultrasound images, acquired in the midsagittal plane.
The dataset consists of 110 patients (P000-P109), each with several video frames. Every frame is annotated with polygons for 8 anatomical structures: pubis, urethra, bladder, vagina, uterus, anus, rectum and levator ani muscle.
This dataset is located at https://doi.org/10.5281/zenodo.10800787 (Zenodo, CC BY 4.0). The dataset is from the publication https://doi.org/10.1016/j.dib.2025.112346. Please cite it if you use this dataset for your research.
1"""PFUS1 is a dataset for segmentation of pelvic floor anatomical structures in transperineal 2ultrasound images, acquired in the midsagittal plane. 3 4The dataset consists of 110 patients (P000-P109), each with several video frames. Every frame is 5annotated with polygons for 8 anatomical structures: pubis, urethra, bladder, vagina, uterus, anus, 6rectum and levator ani muscle. 7 8This dataset is located at https://doi.org/10.5281/zenodo.10800787 (Zenodo, CC BY 4.0). 9The dataset is from the publication https://doi.org/10.1016/j.dib.2025.112346. 10Please cite it if you use this dataset for your research. 11""" 12 13import os 14import json 15from glob import glob 16from tqdm import tqdm 17from natsort import natsorted 18from typing import Union, Tuple, List 19 20import numpy as np 21from PIL import Image 22import imageio.v3 as imageio 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31URL = "https://zenodo.org/records/10800787/files/pfus1.zip" 32CHECKSUM = "e65bd7ca941b895ca90c5891b8866ab78bee4575fc7a7c276b9a655456eea380" 33 34LABEL_MAP = { 35 "Pubis": 1, 36 "Urethra": 2, 37 "Bladder": 3, 38 "Vagina": 4, 39 "Uterus": 5, 40 "Anus": 6, 41 "Rectum": 7, 42 "Levator ani muscle": 8, 43} 44 45 46def get_pfus1_data(path: Union[os.PathLike, str], download: bool = False) -> str: 47 """Download the PFUS1 data. 48 49 Args: 50 path: Filepath to a folder where the data is downloaded for further processing. 51 download: Whether to download the data if it is not present. 52 53 Returns: 54 Filepath where the data is downloaded. 55 """ 56 data_dir = os.path.join(path, "data") 57 if os.path.exists(data_dir): 58 return data_dir 59 60 os.makedirs(path, exist_ok=True) 61 62 zip_path = os.path.join(path, "pfus1.zip") 63 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 64 util.unzip(zip_path=zip_path, dst=path) 65 66 return data_dir 67 68 69def _rasterize_labels(annotation_path, image_path, label_path): 70 if os.path.exists(label_path): 71 return 72 73 with open(annotation_path) as f: 74 annotation = json.load(f) 75 76 from skimage.draw import polygon as draw_polygon 77 78 with Image.open(image_path) as im: 79 shape = (im.height, im.width) 80 81 labels = np.zeros(shape, dtype="uint8") 82 for shape_annotation in annotation: 83 label_id = LABEL_MAP[shape_annotation["label"]] 84 points = np.array(shape_annotation["pol"], dtype=float) 85 rows, columns = draw_polygon(points[:, 1], points[:, 0], shape=shape) 86 labels[rows, columns] = label_id 87 88 os.makedirs(os.path.dirname(label_path), exist_ok=True) 89 imageio.imwrite(label_path, labels, compression="zlib") 90 91 92def get_pfus1_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 93 """Get paths to the PFUS1 data. 94 95 Args: 96 path: Filepath to a folder where the data is downloaded for further processing. 97 download: Whether to download the data if it is not present. 98 99 Returns: 100 List of filepaths for the image data. 101 List of filepaths for the label data. 102 """ 103 data_dir = get_pfus1_data(path, download) 104 105 image_paths = natsorted(glob(os.path.join(data_dir, "P*", "frame_*.png"))) 106 assert len(image_paths) > 0 107 108 label_dir = os.path.join(os.path.dirname(data_dir), "labels") 109 label_paths = [] 110 for image_path in tqdm(image_paths, desc="Rasterize the PFUS1 annotations"): 111 patient_id = os.path.basename(os.path.dirname(image_path)) 112 frame_id = os.path.splitext(os.path.basename(image_path))[0] 113 114 annotation_path = os.path.join(data_dir, patient_id, f"{frame_id}.json") 115 label_path = os.path.join(label_dir, patient_id, f"{frame_id}.tif") 116 117 _rasterize_labels(annotation_path, image_path, label_path) 118 label_paths.append(label_path) 119 120 assert len(image_paths) == len(label_paths) 121 122 return image_paths, label_paths 123 124 125def get_pfus1_dataset( 126 path: Union[os.PathLike, str], 127 patch_shape: Tuple[int, int], 128 resize_inputs: bool = False, 129 download: bool = False, 130 **kwargs 131) -> Dataset: 132 """Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound. 133 134 Args: 135 path: Filepath to a folder where the data is downloaded for further processing. 136 patch_shape: The patch shape to use for training. 137 resize_inputs: Whether to resize the inputs to the patch shape. 138 download: Whether to download the data if it is not present. 139 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 140 141 Returns: 142 The segmentation dataset. 143 """ 144 image_paths, label_paths = get_pfus1_paths(path, download) 145 146 if resize_inputs: 147 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 148 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 149 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 150 ) 151 152 return torch_em.default_segmentation_dataset( 153 raw_paths=image_paths, 154 raw_key=None, 155 label_paths=label_paths, 156 label_key=None, 157 patch_shape=patch_shape, 158 is_seg_dataset=False, 159 **kwargs 160 ) 161 162 163def get_pfus1_loader( 164 path: Union[os.PathLike, str], 165 batch_size: int, 166 patch_shape: Tuple[int, int], 167 resize_inputs: bool = False, 168 download: bool = False, 169 **kwargs 170) -> DataLoader: 171 """Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound. 172 173 Args: 174 path: Filepath to a folder where the data is downloaded for further processing. 175 batch_size: The batch size for training. 176 patch_shape: The patch shape to use for training. 177 resize_inputs: Whether to resize the inputs to the patch shape. 178 download: Whether to download the data if it is not present. 179 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 180 181 Returns: 182 The DataLoader. 183 """ 184 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 185 dataset = get_pfus1_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 186 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
47def get_pfus1_data(path: Union[os.PathLike, str], download: bool = False) -> str: 48 """Download the PFUS1 data. 49 50 Args: 51 path: Filepath to a folder where the data is downloaded for further processing. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 data_dir = os.path.join(path, "data") 58 if os.path.exists(data_dir): 59 return data_dir 60 61 os.makedirs(path, exist_ok=True) 62 63 zip_path = os.path.join(path, "pfus1.zip") 64 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 65 util.unzip(zip_path=zip_path, dst=path) 66 67 return data_dir
Download the PFUS1 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
93def get_pfus1_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 94 """Get paths to the PFUS1 data. 95 96 Args: 97 path: Filepath to a folder where the data is downloaded for further processing. 98 download: Whether to download the data if it is not present. 99 100 Returns: 101 List of filepaths for the image data. 102 List of filepaths for the label data. 103 """ 104 data_dir = get_pfus1_data(path, download) 105 106 image_paths = natsorted(glob(os.path.join(data_dir, "P*", "frame_*.png"))) 107 assert len(image_paths) > 0 108 109 label_dir = os.path.join(os.path.dirname(data_dir), "labels") 110 label_paths = [] 111 for image_path in tqdm(image_paths, desc="Rasterize the PFUS1 annotations"): 112 patient_id = os.path.basename(os.path.dirname(image_path)) 113 frame_id = os.path.splitext(os.path.basename(image_path))[0] 114 115 annotation_path = os.path.join(data_dir, patient_id, f"{frame_id}.json") 116 label_path = os.path.join(label_dir, patient_id, f"{frame_id}.tif") 117 118 _rasterize_labels(annotation_path, image_path, label_path) 119 label_paths.append(label_path) 120 121 assert len(image_paths) == len(label_paths) 122 123 return image_paths, label_paths
Get paths to the PFUS1 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
126def get_pfus1_dataset( 127 path: Union[os.PathLike, str], 128 patch_shape: Tuple[int, int], 129 resize_inputs: bool = False, 130 download: bool = False, 131 **kwargs 132) -> Dataset: 133 """Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound. 134 135 Args: 136 path: Filepath to a folder where the data is downloaded for further processing. 137 patch_shape: The patch shape to use for training. 138 resize_inputs: Whether to resize the inputs to the patch shape. 139 download: Whether to download the data if it is not present. 140 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 141 142 Returns: 143 The segmentation dataset. 144 """ 145 image_paths, label_paths = get_pfus1_paths(path, download) 146 147 if resize_inputs: 148 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 149 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 150 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 151 ) 152 153 return torch_em.default_segmentation_dataset( 154 raw_paths=image_paths, 155 raw_key=None, 156 label_paths=label_paths, 157 label_key=None, 158 patch_shape=patch_shape, 159 is_seg_dataset=False, 160 **kwargs 161 )
Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
164def get_pfus1_loader( 165 path: Union[os.PathLike, str], 166 batch_size: int, 167 patch_shape: Tuple[int, int], 168 resize_inputs: bool = False, 169 download: bool = False, 170 **kwargs 171) -> DataLoader: 172 """Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound. 173 174 Args: 175 path: Filepath to a folder where the data is downloaded for further processing. 176 batch_size: The batch size for training. 177 patch_shape: The patch shape to use for training. 178 resize_inputs: Whether to resize the inputs to the patch shape. 179 download: Whether to download the data if it is not present. 180 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 181 182 Returns: 183 The DataLoader. 184 """ 185 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 186 dataset = get_pfus1_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 187 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.