torch_em.data.datasets.light_microscopy.pcmmd
The PCMMD dataset contains annotations for plasma cell segmentation in bright-field microscopy images of Wright-Giemsa stained bone marrow smears.
The images were captured with a smartphone camera mounted on a microscope at 100x magnification, at the feathered edge of bone marrow aspirate smears from patients investigated for multiple myeloma. The 'segmentation' subset used by this loader holds single-cell crops, each with one polygon annotation for a plasma cell or a non-plasma cell: 1,613 plasma cell crops and 1,927 non-plasma cell crops.
NOTE: The full release also has a 'detection' subset with bounding box annotations for whole slide crops, and per-patient diagnosis data. This loader only prepares the 'segmentation' subset, since it is the only one with pixel-level annotations.
NOTE: The publication defines no split for the 'segmentation' subset. This loader creates a stratified 80/20 train/test split and caches it to disk, so it stays the same across runs.
The dataset is located at https://doi.org/10.17632/3v2nrxpr9s.1 under the CC BY 4.0 license. This dataset is from the publication https://doi.org/10.1038/s41597-025-04459-1. Please cite it if you use this dataset in your research.
1"""The PCMMD dataset contains annotations for plasma cell segmentation in 2bright-field microscopy images of Wright-Giemsa stained bone marrow smears. 3 4The images were captured with a smartphone camera mounted on a microscope at 100x 5magnification, at the feathered edge of bone marrow aspirate smears from patients 6investigated for multiple myeloma. The 'segmentation' subset used by this loader holds 7single-cell crops, each with one polygon annotation for a plasma cell or a non-plasma 8cell: 1,613 plasma cell crops and 1,927 non-plasma cell crops. 9 10NOTE: The full release also has a 'detection' subset with bounding box annotations for 11whole slide crops, and per-patient diagnosis data. This loader only prepares the 12'segmentation' subset, since it is the only one with pixel-level annotations. 13 14NOTE: The publication defines no split for the 'segmentation' subset. This loader creates 15a stratified 80/20 train/test split and caches it to disk, so it stays the same across runs. 16 17The dataset is located at https://doi.org/10.17632/3v2nrxpr9s.1 under the CC BY 4.0 license. 18This dataset is from the publication https://doi.org/10.1038/s41597-025-04459-1. 19Please cite it if you use this dataset in your research. 20""" 21 22import os 23import json 24import zipfile 25from glob import glob 26from pathlib import Path 27from natsort import natsorted 28from typing import List, Literal, Tuple, Union 29 30import numpy as np 31import pandas as pd 32import imageio.v3 as imageio 33 34from torch.utils.data import DataLoader, Dataset 35 36import torch_em 37 38from .. import util 39 40 41URL = "https://data.mendeley.com/public-api/zip/3v2nrxpr9s/download/1" 42CHECKSUM = "a2e3e2367fc12d1e38bfda28137e86a17c256e29ed73817137fe2ca83207da67" 43 44ARCHIVE_FOLDER = "PCMMD Plasma Cells for Multiple Myeloma Diagnosis" 45 46# The dataset ships two cell folders, each with an 'images' and a 'masks' subfolder. 47CELL_FOLDERS = {"plasma": "plasma cells", "non_plasma": "non-plasma cells"} 48 49# The label field in the annotation JSON files marks the class of the segmented cell. 50LABEL_IDS = {"plasma_cell": 1, "non_plasma_cell": 2} 51 52 53def _extract_segmentation_folder(zip_path: str, path: str) -> None: 54 """Extract only the segmentation folder of the archive.""" 55 prefix = f"{ARCHIVE_FOLDER}/data/segmentation/" 56 with zipfile.ZipFile(zip_path) as archive: 57 members = [n for n in archive.namelist() if n.startswith(prefix)] 58 if not members: 59 raise RuntimeError(f"The archive {zip_path} does not hold a 'data/segmentation' folder.") 60 archive.extractall(path, members=members) 61 62 63def _rasterize(shapes, shape: Tuple[int, int]) -> np.ndarray: 64 """Draw the polygon of the segmented cell, with the pixel value set to its class id.""" 65 from skimage.draw import polygon as draw_polygon 66 67 semantic = np.zeros(shape, dtype="uint8") 68 for item in shapes: 69 points = np.array(item["points"], dtype=float) 70 rows, columns = draw_polygon(points[:, 1], points[:, 0], shape=shape) 71 semantic[rows, columns] = LABEL_IDS.get(item.get("label"), 0) 72 return semantic 73 74 75def _create_labels(cell_dir: str) -> str: 76 """Rasterize the polygon of every crop into a label image.""" 77 from tqdm import tqdm 78 79 label_dir = os.path.join(cell_dir, "labels") 80 os.makedirs(label_dir, exist_ok=True) 81 82 json_paths = natsorted(glob(os.path.join(cell_dir, "masks", "*.json"))) 83 for json_path in tqdm(json_paths, desc=f"Preprocess the PCMMD annotations in {cell_dir}"): 84 stem = Path(json_path).stem 85 label_path = os.path.join(label_dir, f"{stem}.tif") 86 if os.path.exists(label_path): 87 continue 88 89 with open(json_path) as f: 90 annotation = json.load(f) 91 92 shape = (annotation["imageHeight"], annotation["imageWidth"]) 93 semantic = _rasterize(annotation.get("shapes", []), shape) 94 imageio.imwrite(label_path, semantic, compression="zlib") 95 96 return label_dir 97 98 99def _create_split_csv(path: str, sample_ids: List[str]) -> pd.DataFrame: 100 from sklearn.model_selection import train_test_split 101 102 csv_path = os.path.join(path, "pcmmd_split.csv") 103 if os.path.exists(csv_path): 104 return pd.read_csv(csv_path) 105 106 print(f"Creating a new split file at '{csv_path}'.") 107 train_ids, test_ids = train_test_split(sample_ids, test_size=0.2, random_state=42) 108 split = {sid: "train" for sid in train_ids} 109 split.update({sid: "test" for sid in test_ids}) 110 df = pd.DataFrame({"sample_id": list(split.keys()), "split": list(split.values())}) 111 df.to_csv(csv_path, index=False) 112 return df 113 114 115def get_pcmmd_data(path: Union[os.PathLike, str], download: bool = False) -> str: 116 """Download the PCMMD dataset. 117 118 Args: 119 path: Filepath to a folder where the downloaded data will be saved. 120 download: Whether to download the data if it is not present. 121 122 Returns: 123 The filepath to the extracted segmentation data. 124 """ 125 data_dir = os.path.join(path, ARCHIVE_FOLDER, "data", "segmentation") 126 if os.path.exists(data_dir): 127 return data_dir 128 129 os.makedirs(path, exist_ok=True) 130 zip_path = os.path.join(path, "pcmmd.zip") 131 util.download_source(zip_path, URL, download, CHECKSUM) 132 _extract_segmentation_folder(zip_path, path) 133 134 return data_dir 135 136 137def get_pcmmd_paths( 138 path: Union[os.PathLike, str], 139 split: Literal["train", "test"] = "train", 140 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 141 download: bool = False, 142) -> Tuple[List[str], List[str]]: 143 """Get paths to the PCMMD data. 144 145 Args: 146 path: Filepath to a folder where the downloaded data will be saved. 147 split: The data split. Either 'train' or 'test'. 148 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 149 download: Whether to download the data if it is not present. 150 151 Returns: 152 List of filepaths for the image data. 153 List of filepaths for the label data. 154 """ 155 if split not in ("train", "test"): 156 raise ValueError(f"'{split}' is not a valid split. Choose from 'train' or 'test'.") 157 if cell_type not in ("plasma", "non_plasma", "both"): 158 raise ValueError(f"'{cell_type}' is not a valid cell type. Choose from 'plasma', 'non_plasma' or 'both'.") 159 160 data_dir = get_pcmmd_data(path, download) 161 cell_types = list(CELL_FOLDERS.keys()) if cell_type == "both" else [cell_type] 162 163 pairs = [] 164 for ctype in cell_types: 165 cell_dir = os.path.join(data_dir, CELL_FOLDERS[ctype]) 166 label_dir = _create_labels(cell_dir) 167 image_paths = natsorted(glob(os.path.join(cell_dir, "images", "*.jpg"))) 168 for image_path in image_paths: 169 stem = Path(image_path).stem 170 label_path = os.path.join(label_dir, f"{stem}.tif") 171 if not os.path.exists(label_path): 172 continue 173 pairs.append((f"{ctype}_{stem}", image_path, label_path)) 174 175 if not pairs: 176 raise RuntimeError(f"Could not find any PCMMD data for cell_type='{cell_type}' in {data_dir}.") 177 178 split_df = _create_split_csv(data_dir, [sample_id for sample_id, _, _ in pairs]) 179 split_ids = set(split_df[split_df["split"] == split]["sample_id"]) 180 181 image_paths = [image_path for sample_id, image_path, _ in pairs if sample_id in split_ids] 182 label_paths = [label_path for sample_id, _, label_path in pairs if sample_id in split_ids] 183 184 if not image_paths: 185 raise RuntimeError(f"Could not find any PCMMD data for split='{split}' in {data_dir}.") 186 187 return image_paths, label_paths 188 189 190def get_pcmmd_dataset( 191 path: Union[os.PathLike, str], 192 patch_shape: Tuple[int, int], 193 split: Literal["train", "test"] = "train", 194 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 195 label_choice: Literal["semantic", "binary"] = "semantic", 196 download: bool = False, 197 **kwargs, 198) -> Dataset: 199 """Get the PCMMD dataset for plasma cell segmentation. 200 201 Args: 202 path: Filepath to a folder where the downloaded data will be saved. 203 patch_shape: The 2D patch shape to use for training. 204 split: The data split. Either 'train' or 'test'. 205 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 206 label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a 207 non-plasma cell 2, or 'binary', where every cell is labeled 1. 208 download: Whether to download the data if it is not present. 209 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 210 211 Returns: 212 The segmentation dataset. 213 """ 214 if len(patch_shape) != 2: 215 raise ValueError(f"The PCMMD patch shape must be two-dimensional, got {patch_shape}.") 216 if label_choice not in ("semantic", "binary"): 217 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose 'semantic' or 'binary'.") 218 219 image_paths, label_paths = get_pcmmd_paths(path, split, cell_type, download) 220 221 if label_choice == "binary": 222 kwargs["label_transform"] = torch_em.transform.label.labels_to_binary 223 kwargs = util.ensure_transforms(ndim=2, **kwargs) 224 225 return torch_em.default_segmentation_dataset( 226 raw_paths=image_paths, 227 raw_key=None, 228 label_paths=label_paths, 229 label_key=None, 230 patch_shape=patch_shape, 231 is_seg_dataset=False, 232 ndim=2, 233 **kwargs, 234 ) 235 236 237def get_pcmmd_loader( 238 path: Union[os.PathLike, str], 239 batch_size: int, 240 patch_shape: Tuple[int, int], 241 split: Literal["train", "test"] = "train", 242 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 243 label_choice: Literal["semantic", "binary"] = "semantic", 244 download: bool = False, 245 **kwargs, 246) -> DataLoader: 247 """Get the PCMMD dataloader for plasma cell segmentation. 248 249 Args: 250 path: Filepath to a folder where the downloaded data will be saved. 251 batch_size: The batch size for training. 252 patch_shape: The 2D patch shape to use for training. 253 split: The data split. Either 'train' or 'test'. 254 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 255 label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a 256 non-plasma cell 2, or 'binary', where every cell is labeled 1. 257 download: Whether to download the data if it is not present. 258 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 259 260 Returns: 261 The DataLoader. 262 """ 263 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 264 dataset = get_pcmmd_dataset( 265 path=path, 266 patch_shape=patch_shape, 267 split=split, 268 cell_type=cell_type, 269 label_choice=label_choice, 270 download=download, 271 **ds_kwargs, 272 ) 273 return torch_em.get_data_loader(dataset, batch_size=batch_size, **loader_kwargs)
116def get_pcmmd_data(path: Union[os.PathLike, str], download: bool = False) -> str: 117 """Download the PCMMD dataset. 118 119 Args: 120 path: Filepath to a folder where the downloaded data will be saved. 121 download: Whether to download the data if it is not present. 122 123 Returns: 124 The filepath to the extracted segmentation data. 125 """ 126 data_dir = os.path.join(path, ARCHIVE_FOLDER, "data", "segmentation") 127 if os.path.exists(data_dir): 128 return data_dir 129 130 os.makedirs(path, exist_ok=True) 131 zip_path = os.path.join(path, "pcmmd.zip") 132 util.download_source(zip_path, URL, download, CHECKSUM) 133 _extract_segmentation_folder(zip_path, path) 134 135 return data_dir
Download the PCMMD dataset.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- download: Whether to download the data if it is not present.
Returns:
The filepath to the extracted segmentation data.
138def get_pcmmd_paths( 139 path: Union[os.PathLike, str], 140 split: Literal["train", "test"] = "train", 141 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 142 download: bool = False, 143) -> Tuple[List[str], List[str]]: 144 """Get paths to the PCMMD data. 145 146 Args: 147 path: Filepath to a folder where the downloaded data will be saved. 148 split: The data split. Either 'train' or 'test'. 149 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 150 download: Whether to download the data if it is not present. 151 152 Returns: 153 List of filepaths for the image data. 154 List of filepaths for the label data. 155 """ 156 if split not in ("train", "test"): 157 raise ValueError(f"'{split}' is not a valid split. Choose from 'train' or 'test'.") 158 if cell_type not in ("plasma", "non_plasma", "both"): 159 raise ValueError(f"'{cell_type}' is not a valid cell type. Choose from 'plasma', 'non_plasma' or 'both'.") 160 161 data_dir = get_pcmmd_data(path, download) 162 cell_types = list(CELL_FOLDERS.keys()) if cell_type == "both" else [cell_type] 163 164 pairs = [] 165 for ctype in cell_types: 166 cell_dir = os.path.join(data_dir, CELL_FOLDERS[ctype]) 167 label_dir = _create_labels(cell_dir) 168 image_paths = natsorted(glob(os.path.join(cell_dir, "images", "*.jpg"))) 169 for image_path in image_paths: 170 stem = Path(image_path).stem 171 label_path = os.path.join(label_dir, f"{stem}.tif") 172 if not os.path.exists(label_path): 173 continue 174 pairs.append((f"{ctype}_{stem}", image_path, label_path)) 175 176 if not pairs: 177 raise RuntimeError(f"Could not find any PCMMD data for cell_type='{cell_type}' in {data_dir}.") 178 179 split_df = _create_split_csv(data_dir, [sample_id for sample_id, _, _ in pairs]) 180 split_ids = set(split_df[split_df["split"] == split]["sample_id"]) 181 182 image_paths = [image_path for sample_id, image_path, _ in pairs if sample_id in split_ids] 183 label_paths = [label_path for sample_id, _, label_path in pairs if sample_id in split_ids] 184 185 if not image_paths: 186 raise RuntimeError(f"Could not find any PCMMD data for split='{split}' in {data_dir}.") 187 188 return image_paths, label_paths
Get paths to the PCMMD data.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- split: The data split. Either 'train' or 'test'.
- cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
191def get_pcmmd_dataset( 192 path: Union[os.PathLike, str], 193 patch_shape: Tuple[int, int], 194 split: Literal["train", "test"] = "train", 195 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 196 label_choice: Literal["semantic", "binary"] = "semantic", 197 download: bool = False, 198 **kwargs, 199) -> Dataset: 200 """Get the PCMMD dataset for plasma cell segmentation. 201 202 Args: 203 path: Filepath to a folder where the downloaded data will be saved. 204 patch_shape: The 2D patch shape to use for training. 205 split: The data split. Either 'train' or 'test'. 206 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 207 label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a 208 non-plasma cell 2, or 'binary', where every cell is labeled 1. 209 download: Whether to download the data if it is not present. 210 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 211 212 Returns: 213 The segmentation dataset. 214 """ 215 if len(patch_shape) != 2: 216 raise ValueError(f"The PCMMD patch shape must be two-dimensional, got {patch_shape}.") 217 if label_choice not in ("semantic", "binary"): 218 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose 'semantic' or 'binary'.") 219 220 image_paths, label_paths = get_pcmmd_paths(path, split, cell_type, download) 221 222 if label_choice == "binary": 223 kwargs["label_transform"] = torch_em.transform.label.labels_to_binary 224 kwargs = util.ensure_transforms(ndim=2, **kwargs) 225 226 return torch_em.default_segmentation_dataset( 227 raw_paths=image_paths, 228 raw_key=None, 229 label_paths=label_paths, 230 label_key=None, 231 patch_shape=patch_shape, 232 is_seg_dataset=False, 233 ndim=2, 234 **kwargs, 235 )
Get the PCMMD dataset for plasma cell segmentation.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- patch_shape: The 2D patch shape to use for training.
- split: The data split. Either 'train' or 'test'.
- cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
- label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a non-plasma cell 2, or 'binary', where every cell is labeled 1.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
238def get_pcmmd_loader( 239 path: Union[os.PathLike, str], 240 batch_size: int, 241 patch_shape: Tuple[int, int], 242 split: Literal["train", "test"] = "train", 243 cell_type: Literal["plasma", "non_plasma", "both"] = "both", 244 label_choice: Literal["semantic", "binary"] = "semantic", 245 download: bool = False, 246 **kwargs, 247) -> DataLoader: 248 """Get the PCMMD dataloader for plasma cell segmentation. 249 250 Args: 251 path: Filepath to a folder where the downloaded data will be saved. 252 batch_size: The batch size for training. 253 patch_shape: The 2D patch shape to use for training. 254 split: The data split. Either 'train' or 'test'. 255 cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'. 256 label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a 257 non-plasma cell 2, or 'binary', where every cell is labeled 1. 258 download: Whether to download the data if it is not present. 259 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 260 261 Returns: 262 The DataLoader. 263 """ 264 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 265 dataset = get_pcmmd_dataset( 266 path=path, 267 patch_shape=patch_shape, 268 split=split, 269 cell_type=cell_type, 270 label_choice=label_choice, 271 download=download, 272 **ds_kwargs, 273 ) 274 return torch_em.get_data_loader(dataset, batch_size=batch_size, **loader_kwargs)
Get the PCMMD dataloader for plasma cell segmentation.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- batch_size: The batch size for training.
- patch_shape: The 2D patch shape to use for training.
- split: The data split. Either 'train' or 'test'.
- cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
- label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a non-plasma cell 2, or 'binary', where every cell is labeled 1.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor the PyTorch DataLoader.
Returns:
The DataLoader.