torch_em.data.datasets.medical.fetoplac
The Fetoscopy Placenta Dataset contains annotations for placental vessel segmentation in in-vivo fetoscopic videos of twin-to-twin transfusion syndrome surgery.
It consists of 483 frames with binary vessel masks, drawn from 6 annotated video clips (6 further clips are provided unannotated, for mosaicking, and are not used by this module).
NOTE: The original host (weiss-develop.cs.ucl.ac.uk, UCL) has become unreachable; this module downloads the same file from the Internet Archive's Wayback Machine, which mirrors it unchanged.
The dataset is located at https://www.ucl.ac.uk/interventional-surgical-sciences/fetoscopy-placenta-data.
This dataset is from the publication https://doi.org/10.1007/978-3-030-59716-0_73. Please cite it if you use this dataset for your research.
1"""The Fetoscopy Placenta Dataset contains annotations for placental vessel segmentation 2in in-vivo fetoscopic videos of twin-to-twin transfusion syndrome surgery. 3 4It consists of 483 frames with binary vessel masks, drawn from 6 annotated video clips 5(6 further clips are provided unannotated, for mosaicking, and are not used by this module). 6 7NOTE: The original host (weiss-develop.cs.ucl.ac.uk, UCL) has become unreachable; this module 8downloads the same file from the Internet Archive's Wayback Machine, which mirrors it unchanged. 9 10The dataset is located at https://www.ucl.ac.uk/interventional-surgical-sciences/fetoscopy-placenta-data. 11 12This dataset is from the publication https://doi.org/10.1007/978-3-030-59716-0_73. 13Please cite it if you use this dataset for your research. 14""" 15 16import os 17from glob import glob 18from tqdm import tqdm 19from pathlib import Path 20from typing import Union, Tuple, List 21 22import numpy as np 23import imageio.v3 as imageio 24 25from torch.utils.data import Dataset, DataLoader 26 27import torch_em 28 29from .. import util 30 31 32URL = "http://web.archive.org/web/20220412011703/https://weiss-develop.cs.ucl.ac.uk/fetoscopy-data/fetoscopy-placenta-dataset/fetoscopy-placenta-dataset.zip" # noqa 33CHECKSUM = "a7beadf24f377d80c2305b10139697eee2ecce4f3ac8564dcec604c5f2dc55df" 34 35 36def get_fetoplac_data(path: Union[os.PathLike, str], download: bool = False) -> str: 37 """Download the Fetoscopy Placenta Dataset. 38 39 Args: 40 path: Filepath to a folder where the data is downloaded for further processing. 41 download: Whether to download the data if it is not present. 42 43 Returns: 44 Filepath to the folder with the annotated vessel segmentation videos. 45 """ 46 data_dir = os.path.join(path, "Fetoscopy Placenta Dataset", "Vessel_segmentation_annotations") 47 if os.path.exists(data_dir): 48 return data_dir 49 50 os.makedirs(path, exist_ok=True) 51 52 zip_path = os.path.join(path, "fetoscopy-placenta-dataset.zip") 53 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 54 util.unzip(zip_path=zip_path, dst=path) 55 56 return data_dir 57 58 59def get_fetoplac_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 60 """Get paths to the Fetoscopy Placenta Dataset data. 61 62 Args: 63 path: Filepath to a folder where the data is downloaded for further processing. 64 download: Whether to download the data if it is not present. 65 66 Returns: 67 List of filepaths for the image data. 68 List of filepaths for the label data. 69 """ 70 data_dir = get_fetoplac_data(path=path, download=download) 71 72 image_paths = sorted(glob(os.path.join(data_dir, "video*", "images", "*.png"))) 73 gt_paths = sorted(glob(os.path.join(data_dir, "video*", "masks_gt", "*_mask.png"))) 74 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, "Could not find matching image/mask pairs." 75 76 neu_gt_paths = [] 77 for gt_path in tqdm(gt_paths, desc="Preprocessing Fetoscopy Placenta Dataset masks"): 78 neu_gt_path = os.path.join(Path(gt_path).parent, f"{Path(gt_path).stem}_binary.tif") 79 neu_gt_paths.append(neu_gt_path) 80 if os.path.exists(neu_gt_path): 81 continue 82 83 gt = imageio.imread(gt_path) 84 gt = np.mean(gt, axis=-1) 85 gt = (gt > 0).astype("uint8") 86 imageio.imwrite(neu_gt_path, gt, compression="zlib") 87 88 return image_paths, neu_gt_paths 89 90 91def get_fetoplac_dataset( 92 path: Union[os.PathLike, str], 93 patch_shape: Tuple[int, int], 94 resize_inputs: bool = False, 95 download: bool = False, 96 **kwargs 97) -> Dataset: 98 """Get the Fetoscopy Placenta Dataset for placental vessel segmentation. 99 100 Args: 101 path: Filepath to a folder where the data is downloaded for further processing. 102 patch_shape: The patch shape to use for training. 103 resize_inputs: Whether to resize the inputs to the patch shape. 104 download: Whether to download the data if it is not present. 105 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 106 107 Returns: 108 The segmentation dataset. 109 """ 110 image_paths, gt_paths = get_fetoplac_paths(path, download) 111 112 if resize_inputs: 113 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 114 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 115 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 116 ) 117 118 return torch_em.default_segmentation_dataset( 119 raw_paths=image_paths, 120 raw_key=None, 121 label_paths=gt_paths, 122 label_key=None, 123 patch_shape=patch_shape, 124 is_seg_dataset=False, 125 **kwargs 126 ) 127 128 129def get_fetoplac_loader( 130 path: Union[os.PathLike, str], 131 patch_shape: Tuple[int, int], 132 batch_size: int, 133 resize_inputs: bool = False, 134 download: bool = False, 135 **kwargs 136) -> DataLoader: 137 """Get the Fetoscopy Placenta Dataset dataloader for placental vessel segmentation. 138 139 Args: 140 path: Filepath to a folder where the data is downloaded for further processing. 141 patch_shape: The patch shape to use for training. 142 batch_size: The batch size for training. 143 resize_inputs: Whether to resize the inputs to the patch shape. 144 download: Whether to download the data if it is not present. 145 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 146 147 Returns: 148 The DataLoader. 149 """ 150 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 151 dataset = get_fetoplac_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 152 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
37def get_fetoplac_data(path: Union[os.PathLike, str], download: bool = False) -> str: 38 """Download the Fetoscopy Placenta Dataset. 39 40 Args: 41 path: Filepath to a folder where the data is downloaded for further processing. 42 download: Whether to download the data if it is not present. 43 44 Returns: 45 Filepath to the folder with the annotated vessel segmentation videos. 46 """ 47 data_dir = os.path.join(path, "Fetoscopy Placenta Dataset", "Vessel_segmentation_annotations") 48 if os.path.exists(data_dir): 49 return data_dir 50 51 os.makedirs(path, exist_ok=True) 52 53 zip_path = os.path.join(path, "fetoscopy-placenta-dataset.zip") 54 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 55 util.unzip(zip_path=zip_path, dst=path) 56 57 return data_dir
Download the Fetoscopy Placenta Dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder with the annotated vessel segmentation videos.
60def get_fetoplac_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 61 """Get paths to the Fetoscopy Placenta Dataset data. 62 63 Args: 64 path: Filepath to a folder where the data is downloaded for further processing. 65 download: Whether to download the data if it is not present. 66 67 Returns: 68 List of filepaths for the image data. 69 List of filepaths for the label data. 70 """ 71 data_dir = get_fetoplac_data(path=path, download=download) 72 73 image_paths = sorted(glob(os.path.join(data_dir, "video*", "images", "*.png"))) 74 gt_paths = sorted(glob(os.path.join(data_dir, "video*", "masks_gt", "*_mask.png"))) 75 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, "Could not find matching image/mask pairs." 76 77 neu_gt_paths = [] 78 for gt_path in tqdm(gt_paths, desc="Preprocessing Fetoscopy Placenta Dataset masks"): 79 neu_gt_path = os.path.join(Path(gt_path).parent, f"{Path(gt_path).stem}_binary.tif") 80 neu_gt_paths.append(neu_gt_path) 81 if os.path.exists(neu_gt_path): 82 continue 83 84 gt = imageio.imread(gt_path) 85 gt = np.mean(gt, axis=-1) 86 gt = (gt > 0).astype("uint8") 87 imageio.imwrite(neu_gt_path, gt, compression="zlib") 88 89 return image_paths, neu_gt_paths
Get paths to the Fetoscopy Placenta Dataset data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
92def get_fetoplac_dataset( 93 path: Union[os.PathLike, str], 94 patch_shape: Tuple[int, int], 95 resize_inputs: bool = False, 96 download: bool = False, 97 **kwargs 98) -> Dataset: 99 """Get the Fetoscopy Placenta Dataset for placental vessel segmentation. 100 101 Args: 102 path: Filepath to a folder where the data is downloaded for further processing. 103 patch_shape: The patch shape to use for training. 104 resize_inputs: Whether to resize the inputs to the patch shape. 105 download: Whether to download the data if it is not present. 106 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 107 108 Returns: 109 The segmentation dataset. 110 """ 111 image_paths, gt_paths = get_fetoplac_paths(path, download) 112 113 if resize_inputs: 114 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 115 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 116 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 117 ) 118 119 return torch_em.default_segmentation_dataset( 120 raw_paths=image_paths, 121 raw_key=None, 122 label_paths=gt_paths, 123 label_key=None, 124 patch_shape=patch_shape, 125 is_seg_dataset=False, 126 **kwargs 127 )
Get the Fetoscopy Placenta Dataset for placental vessel segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
130def get_fetoplac_loader( 131 path: Union[os.PathLike, str], 132 patch_shape: Tuple[int, int], 133 batch_size: int, 134 resize_inputs: bool = False, 135 download: bool = False, 136 **kwargs 137) -> DataLoader: 138 """Get the Fetoscopy Placenta Dataset dataloader for placental vessel segmentation. 139 140 Args: 141 path: Filepath to a folder where the data is downloaded for further processing. 142 patch_shape: The patch shape to use for training. 143 batch_size: The batch size for training. 144 resize_inputs: Whether to resize the inputs to the patch shape. 145 download: Whether to download the data if it is not present. 146 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 147 148 Returns: 149 The DataLoader. 150 """ 151 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 152 dataset = get_fetoplac_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 153 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
Get the Fetoscopy Placenta Dataset dataloader for placental vessel segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- batch_size: The batch size for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.