torch_em.data.datasets.light_microscopy.pcmmd

The PCMMD dataset contains annotations for plasma cell segmentation in bright-field microscopy images of Wright-Giemsa stained bone marrow smears.

The images were captured with a smartphone camera mounted on a microscope at 100x magnification, at the feathered edge of bone marrow aspirate smears from patients investigated for multiple myeloma. The 'segmentation' subset used by this loader holds single-cell crops, each with one polygon annotation for a plasma cell or a non-plasma cell: 1,613 plasma cell crops and 1,927 non-plasma cell crops.

NOTE: The full release also has a 'detection' subset with bounding box annotations for whole slide crops, and per-patient diagnosis data. This loader only prepares the 'segmentation' subset, since it is the only one with pixel-level annotations.

NOTE: The publication defines no split for the 'segmentation' subset. This loader creates a stratified 80/20 train/test split and caches it to disk, so it stays the same across runs.

The dataset is located at https://doi.org/10.17632/3v2nrxpr9s.1 under the CC BY 4.0 license. This dataset is from the publication https://doi.org/10.1038/s41597-025-04459-1. Please cite it if you use this dataset in your research.

  1"""The PCMMD dataset contains annotations for plasma cell segmentation in
  2bright-field microscopy images of Wright-Giemsa stained bone marrow smears.
  3
  4The images were captured with a smartphone camera mounted on a microscope at 100x
  5magnification, at the feathered edge of bone marrow aspirate smears from patients
  6investigated for multiple myeloma. The 'segmentation' subset used by this loader holds
  7single-cell crops, each with one polygon annotation for a plasma cell or a non-plasma
  8cell: 1,613 plasma cell crops and 1,927 non-plasma cell crops.
  9
 10NOTE: The full release also has a 'detection' subset with bounding box annotations for
 11whole slide crops, and per-patient diagnosis data. This loader only prepares the
 12'segmentation' subset, since it is the only one with pixel-level annotations.
 13
 14NOTE: The publication defines no split for the 'segmentation' subset. This loader creates
 15a stratified 80/20 train/test split and caches it to disk, so it stays the same across runs.
 16
 17The dataset is located at https://doi.org/10.17632/3v2nrxpr9s.1 under the CC BY 4.0 license.
 18This dataset is from the publication https://doi.org/10.1038/s41597-025-04459-1.
 19Please cite it if you use this dataset in your research.
 20"""
 21
 22import os
 23import json
 24import zipfile
 25from glob import glob
 26from pathlib import Path
 27from natsort import natsorted
 28from typing import List, Literal, Tuple, Union
 29
 30import numpy as np
 31import pandas as pd
 32import imageio.v3 as imageio
 33
 34from torch.utils.data import DataLoader, Dataset
 35
 36import torch_em
 37
 38from .. import util
 39
 40
 41URL = "https://data.mendeley.com/public-api/zip/3v2nrxpr9s/download/1"
 42CHECKSUM = "a2e3e2367fc12d1e38bfda28137e86a17c256e29ed73817137fe2ca83207da67"
 43
 44ARCHIVE_FOLDER = "PCMMD Plasma Cells for Multiple Myeloma Diagnosis"
 45
 46# The dataset ships two cell folders, each with an 'images' and a 'masks' subfolder.
 47CELL_FOLDERS = {"plasma": "plasma cells", "non_plasma": "non-plasma cells"}
 48
 49# The label field in the annotation JSON files marks the class of the segmented cell.
 50LABEL_IDS = {"plasma_cell": 1, "non_plasma_cell": 2}
 51
 52
 53def _extract_segmentation_folder(zip_path: str, path: str) -> None:
 54    """Extract only the segmentation folder of the archive."""
 55    prefix = f"{ARCHIVE_FOLDER}/data/segmentation/"
 56    with zipfile.ZipFile(zip_path) as archive:
 57        members = [n for n in archive.namelist() if n.startswith(prefix)]
 58        if not members:
 59            raise RuntimeError(f"The archive {zip_path} does not hold a 'data/segmentation' folder.")
 60        archive.extractall(path, members=members)
 61
 62
 63def _rasterize(shapes, shape: Tuple[int, int]) -> np.ndarray:
 64    """Draw the polygon of the segmented cell, with the pixel value set to its class id."""
 65    from skimage.draw import polygon as draw_polygon
 66
 67    semantic = np.zeros(shape, dtype="uint8")
 68    for item in shapes:
 69        points = np.array(item["points"], dtype=float)
 70        rows, columns = draw_polygon(points[:, 1], points[:, 0], shape=shape)
 71        semantic[rows, columns] = LABEL_IDS.get(item.get("label"), 0)
 72    return semantic
 73
 74
 75def _create_labels(cell_dir: str) -> str:
 76    """Rasterize the polygon of every crop into a label image."""
 77    from tqdm import tqdm
 78
 79    label_dir = os.path.join(cell_dir, "labels")
 80    os.makedirs(label_dir, exist_ok=True)
 81
 82    json_paths = natsorted(glob(os.path.join(cell_dir, "masks", "*.json")))
 83    for json_path in tqdm(json_paths, desc=f"Preprocess the PCMMD annotations in {cell_dir}"):
 84        stem = Path(json_path).stem
 85        label_path = os.path.join(label_dir, f"{stem}.tif")
 86        if os.path.exists(label_path):
 87            continue
 88
 89        with open(json_path) as f:
 90            annotation = json.load(f)
 91
 92        shape = (annotation["imageHeight"], annotation["imageWidth"])
 93        semantic = _rasterize(annotation.get("shapes", []), shape)
 94        imageio.imwrite(label_path, semantic, compression="zlib")
 95
 96    return label_dir
 97
 98
 99def _create_split_csv(path: str, sample_ids: List[str]) -> pd.DataFrame:
100    from sklearn.model_selection import train_test_split
101
102    csv_path = os.path.join(path, "pcmmd_split.csv")
103    if os.path.exists(csv_path):
104        return pd.read_csv(csv_path)
105
106    print(f"Creating a new split file at '{csv_path}'.")
107    train_ids, test_ids = train_test_split(sample_ids, test_size=0.2, random_state=42)
108    split = {sid: "train" for sid in train_ids}
109    split.update({sid: "test" for sid in test_ids})
110    df = pd.DataFrame({"sample_id": list(split.keys()), "split": list(split.values())})
111    df.to_csv(csv_path, index=False)
112    return df
113
114
115def get_pcmmd_data(path: Union[os.PathLike, str], download: bool = False) -> str:
116    """Download the PCMMD dataset.
117
118    Args:
119        path: Filepath to a folder where the downloaded data will be saved.
120        download: Whether to download the data if it is not present.
121
122    Returns:
123        The filepath to the extracted segmentation data.
124    """
125    data_dir = os.path.join(path, ARCHIVE_FOLDER, "data", "segmentation")
126    if os.path.exists(data_dir):
127        return data_dir
128
129    os.makedirs(path, exist_ok=True)
130    zip_path = os.path.join(path, "pcmmd.zip")
131    util.download_source(zip_path, URL, download, CHECKSUM)
132    _extract_segmentation_folder(zip_path, path)
133
134    return data_dir
135
136
137def get_pcmmd_paths(
138    path: Union[os.PathLike, str],
139    split: Literal["train", "test"] = "train",
140    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
141    download: bool = False,
142) -> Tuple[List[str], List[str]]:
143    """Get paths to the PCMMD data.
144
145    Args:
146        path: Filepath to a folder where the downloaded data will be saved.
147        split: The data split. Either 'train' or 'test'.
148        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
149        download: Whether to download the data if it is not present.
150
151    Returns:
152        List of filepaths for the image data.
153        List of filepaths for the label data.
154    """
155    if split not in ("train", "test"):
156        raise ValueError(f"'{split}' is not a valid split. Choose from 'train' or 'test'.")
157    if cell_type not in ("plasma", "non_plasma", "both"):
158        raise ValueError(f"'{cell_type}' is not a valid cell type. Choose from 'plasma', 'non_plasma' or 'both'.")
159
160    data_dir = get_pcmmd_data(path, download)
161    cell_types = list(CELL_FOLDERS.keys()) if cell_type == "both" else [cell_type]
162
163    pairs = []
164    for ctype in cell_types:
165        cell_dir = os.path.join(data_dir, CELL_FOLDERS[ctype])
166        label_dir = _create_labels(cell_dir)
167        image_paths = natsorted(glob(os.path.join(cell_dir, "images", "*.jpg")))
168        for image_path in image_paths:
169            stem = Path(image_path).stem
170            label_path = os.path.join(label_dir, f"{stem}.tif")
171            if not os.path.exists(label_path):
172                continue
173            pairs.append((f"{ctype}_{stem}", image_path, label_path))
174
175    if not pairs:
176        raise RuntimeError(f"Could not find any PCMMD data for cell_type='{cell_type}' in {data_dir}.")
177
178    split_df = _create_split_csv(data_dir, [sample_id for sample_id, _, _ in pairs])
179    split_ids = set(split_df[split_df["split"] == split]["sample_id"])
180
181    image_paths = [image_path for sample_id, image_path, _ in pairs if sample_id in split_ids]
182    label_paths = [label_path for sample_id, _, label_path in pairs if sample_id in split_ids]
183
184    if not image_paths:
185        raise RuntimeError(f"Could not find any PCMMD data for split='{split}' in {data_dir}.")
186
187    return image_paths, label_paths
188
189
190def get_pcmmd_dataset(
191    path: Union[os.PathLike, str],
192    patch_shape: Tuple[int, int],
193    split: Literal["train", "test"] = "train",
194    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
195    label_choice: Literal["semantic", "binary"] = "semantic",
196    download: bool = False,
197    **kwargs,
198) -> Dataset:
199    """Get the PCMMD dataset for plasma cell segmentation.
200
201    Args:
202        path: Filepath to a folder where the downloaded data will be saved.
203        patch_shape: The 2D patch shape to use for training.
204        split: The data split. Either 'train' or 'test'.
205        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
206        label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a
207            non-plasma cell 2, or 'binary', where every cell is labeled 1.
208        download: Whether to download the data if it is not present.
209        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
210
211    Returns:
212        The segmentation dataset.
213    """
214    if len(patch_shape) != 2:
215        raise ValueError(f"The PCMMD patch shape must be two-dimensional, got {patch_shape}.")
216    if label_choice not in ("semantic", "binary"):
217        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose 'semantic' or 'binary'.")
218
219    image_paths, label_paths = get_pcmmd_paths(path, split, cell_type, download)
220
221    if label_choice == "binary":
222        kwargs["label_transform"] = torch_em.transform.label.labels_to_binary
223    kwargs = util.ensure_transforms(ndim=2, **kwargs)
224
225    return torch_em.default_segmentation_dataset(
226        raw_paths=image_paths,
227        raw_key=None,
228        label_paths=label_paths,
229        label_key=None,
230        patch_shape=patch_shape,
231        is_seg_dataset=False,
232        ndim=2,
233        **kwargs,
234    )
235
236
237def get_pcmmd_loader(
238    path: Union[os.PathLike, str],
239    batch_size: int,
240    patch_shape: Tuple[int, int],
241    split: Literal["train", "test"] = "train",
242    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
243    label_choice: Literal["semantic", "binary"] = "semantic",
244    download: bool = False,
245    **kwargs,
246) -> DataLoader:
247    """Get the PCMMD dataloader for plasma cell segmentation.
248
249    Args:
250        path: Filepath to a folder where the downloaded data will be saved.
251        batch_size: The batch size for training.
252        patch_shape: The 2D patch shape to use for training.
253        split: The data split. Either 'train' or 'test'.
254        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
255        label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a
256            non-plasma cell 2, or 'binary', where every cell is labeled 1.
257        download: Whether to download the data if it is not present.
258        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
259
260    Returns:
261        The DataLoader.
262    """
263    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
264    dataset = get_pcmmd_dataset(
265        path=path,
266        patch_shape=patch_shape,
267        split=split,
268        cell_type=cell_type,
269        label_choice=label_choice,
270        download=download,
271        **ds_kwargs,
272    )
273    return torch_em.get_data_loader(dataset, batch_size=batch_size, **loader_kwargs)
URL = 'https://data.mendeley.com/public-api/zip/3v2nrxpr9s/download/1'
CHECKSUM = 'a2e3e2367fc12d1e38bfda28137e86a17c256e29ed73817137fe2ca83207da67'
ARCHIVE_FOLDER = 'PCMMD Plasma Cells for Multiple Myeloma Diagnosis'
CELL_FOLDERS = {'plasma': 'plasma cells', 'non_plasma': 'non-plasma cells'}
LABEL_IDS = {'plasma_cell': 1, 'non_plasma_cell': 2}
def get_pcmmd_data(path: Union[os.PathLike, str], download: bool = False) -> str:
116def get_pcmmd_data(path: Union[os.PathLike, str], download: bool = False) -> str:
117    """Download the PCMMD dataset.
118
119    Args:
120        path: Filepath to a folder where the downloaded data will be saved.
121        download: Whether to download the data if it is not present.
122
123    Returns:
124        The filepath to the extracted segmentation data.
125    """
126    data_dir = os.path.join(path, ARCHIVE_FOLDER, "data", "segmentation")
127    if os.path.exists(data_dir):
128        return data_dir
129
130    os.makedirs(path, exist_ok=True)
131    zip_path = os.path.join(path, "pcmmd.zip")
132    util.download_source(zip_path, URL, download, CHECKSUM)
133    _extract_segmentation_folder(zip_path, path)
134
135    return data_dir

Download the PCMMD dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • download: Whether to download the data if it is not present.
Returns:

The filepath to the extracted segmentation data.

def get_pcmmd_paths( path: Union[os.PathLike, str], split: Literal['train', 'test'] = 'train', cell_type: Literal['plasma', 'non_plasma', 'both'] = 'both', download: bool = False) -> Tuple[List[str], List[str]]:
138def get_pcmmd_paths(
139    path: Union[os.PathLike, str],
140    split: Literal["train", "test"] = "train",
141    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
142    download: bool = False,
143) -> Tuple[List[str], List[str]]:
144    """Get paths to the PCMMD data.
145
146    Args:
147        path: Filepath to a folder where the downloaded data will be saved.
148        split: The data split. Either 'train' or 'test'.
149        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
150        download: Whether to download the data if it is not present.
151
152    Returns:
153        List of filepaths for the image data.
154        List of filepaths for the label data.
155    """
156    if split not in ("train", "test"):
157        raise ValueError(f"'{split}' is not a valid split. Choose from 'train' or 'test'.")
158    if cell_type not in ("plasma", "non_plasma", "both"):
159        raise ValueError(f"'{cell_type}' is not a valid cell type. Choose from 'plasma', 'non_plasma' or 'both'.")
160
161    data_dir = get_pcmmd_data(path, download)
162    cell_types = list(CELL_FOLDERS.keys()) if cell_type == "both" else [cell_type]
163
164    pairs = []
165    for ctype in cell_types:
166        cell_dir = os.path.join(data_dir, CELL_FOLDERS[ctype])
167        label_dir = _create_labels(cell_dir)
168        image_paths = natsorted(glob(os.path.join(cell_dir, "images", "*.jpg")))
169        for image_path in image_paths:
170            stem = Path(image_path).stem
171            label_path = os.path.join(label_dir, f"{stem}.tif")
172            if not os.path.exists(label_path):
173                continue
174            pairs.append((f"{ctype}_{stem}", image_path, label_path))
175
176    if not pairs:
177        raise RuntimeError(f"Could not find any PCMMD data for cell_type='{cell_type}' in {data_dir}.")
178
179    split_df = _create_split_csv(data_dir, [sample_id for sample_id, _, _ in pairs])
180    split_ids = set(split_df[split_df["split"] == split]["sample_id"])
181
182    image_paths = [image_path for sample_id, image_path, _ in pairs if sample_id in split_ids]
183    label_paths = [label_path for sample_id, _, label_path in pairs if sample_id in split_ids]
184
185    if not image_paths:
186        raise RuntimeError(f"Could not find any PCMMD data for split='{split}' in {data_dir}.")
187
188    return image_paths, label_paths

Get paths to the PCMMD data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • split: The data split. Either 'train' or 'test'.
  • cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_pcmmd_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'test'] = 'train', cell_type: Literal['plasma', 'non_plasma', 'both'] = 'both', label_choice: Literal['semantic', 'binary'] = 'semantic', download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
191def get_pcmmd_dataset(
192    path: Union[os.PathLike, str],
193    patch_shape: Tuple[int, int],
194    split: Literal["train", "test"] = "train",
195    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
196    label_choice: Literal["semantic", "binary"] = "semantic",
197    download: bool = False,
198    **kwargs,
199) -> Dataset:
200    """Get the PCMMD dataset for plasma cell segmentation.
201
202    Args:
203        path: Filepath to a folder where the downloaded data will be saved.
204        patch_shape: The 2D patch shape to use for training.
205        split: The data split. Either 'train' or 'test'.
206        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
207        label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a
208            non-plasma cell 2, or 'binary', where every cell is labeled 1.
209        download: Whether to download the data if it is not present.
210        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
211
212    Returns:
213        The segmentation dataset.
214    """
215    if len(patch_shape) != 2:
216        raise ValueError(f"The PCMMD patch shape must be two-dimensional, got {patch_shape}.")
217    if label_choice not in ("semantic", "binary"):
218        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose 'semantic' or 'binary'.")
219
220    image_paths, label_paths = get_pcmmd_paths(path, split, cell_type, download)
221
222    if label_choice == "binary":
223        kwargs["label_transform"] = torch_em.transform.label.labels_to_binary
224    kwargs = util.ensure_transforms(ndim=2, **kwargs)
225
226    return torch_em.default_segmentation_dataset(
227        raw_paths=image_paths,
228        raw_key=None,
229        label_paths=label_paths,
230        label_key=None,
231        patch_shape=patch_shape,
232        is_seg_dataset=False,
233        ndim=2,
234        **kwargs,
235    )

Get the PCMMD dataset for plasma cell segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The 2D patch shape to use for training.
  • split: The data split. Either 'train' or 'test'.
  • cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
  • label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a non-plasma cell 2, or 'binary', where every cell is labeled 1.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pcmmd_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'test'] = 'train', cell_type: Literal['plasma', 'non_plasma', 'both'] = 'both', label_choice: Literal['semantic', 'binary'] = 'semantic', download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
238def get_pcmmd_loader(
239    path: Union[os.PathLike, str],
240    batch_size: int,
241    patch_shape: Tuple[int, int],
242    split: Literal["train", "test"] = "train",
243    cell_type: Literal["plasma", "non_plasma", "both"] = "both",
244    label_choice: Literal["semantic", "binary"] = "semantic",
245    download: bool = False,
246    **kwargs,
247) -> DataLoader:
248    """Get the PCMMD dataloader for plasma cell segmentation.
249
250    Args:
251        path: Filepath to a folder where the downloaded data will be saved.
252        batch_size: The batch size for training.
253        patch_shape: The 2D patch shape to use for training.
254        split: The data split. Either 'train' or 'test'.
255        cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
256        label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a
257            non-plasma cell 2, or 'binary', where every cell is labeled 1.
258        download: Whether to download the data if it is not present.
259        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
260
261    Returns:
262        The DataLoader.
263    """
264    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
265    dataset = get_pcmmd_dataset(
266        path=path,
267        patch_shape=patch_shape,
268        split=split,
269        cell_type=cell_type,
270        label_choice=label_choice,
271        download=download,
272        **ds_kwargs,
273    )
274    return torch_em.get_data_loader(dataset, batch_size=batch_size, **loader_kwargs)

Get the PCMMD dataloader for plasma cell segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The 2D patch shape to use for training.
  • split: The data split. Either 'train' or 'test'.
  • cell_type: The cell type to load. One of 'plasma', 'non_plasma' or 'both'.
  • label_choice: The label to use. Either 'semantic', where a plasma cell is labeled 1 and a non-plasma cell 2, or 'binary', where every cell is labeled 1.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or the PyTorch DataLoader.
Returns:

The DataLoader.