torch_em.data.datasets.light_microscopy.bbbc010
The BBBC010 dataset contains brightfield and GFP images of the roundworm C. elegans from a live/dead assay screen for novel anti-infectives. Animals were exposed to the pathogen Enterococcus faecalis and either treated with ampicillin ("live" phenotype) or left untreated ("dead" phenotype).
The dataset contains 100 wells from a 384-well plate, each imaged in two channels (w1: brightfield, w2: GFP), with instance segmentation ground truth for individual worms.
The dataset is located at https://bbbc.broadinstitute.org/BBBC010. This dataset is CC0 (public domain). If you use it, please cite it as: "We used the C. elegans infection live/dead image set version 1 provided by Fred Ausubel and available from the Broad Bioimage Benchmark Collection [Ljosa et al., Nature Methods, 2012]." https://doi.org/10.1038/nmeth.2083
1"""The BBBC010 dataset contains brightfield and GFP images of the roundworm 2C. elegans from a live/dead assay screen for novel anti-infectives. Animals were 3exposed to the pathogen Enterococcus faecalis and either treated with ampicillin 4("live" phenotype) or left untreated ("dead" phenotype). 5 6The dataset contains 100 wells from a 384-well plate, each imaged in two channels 7(w1: brightfield, w2: GFP), with instance segmentation ground truth for individual 8worms. 9 10The dataset is located at https://bbbc.broadinstitute.org/BBBC010. 11This dataset is CC0 (public domain). If you use it, please cite it as: 12"We used the C. elegans infection live/dead image set version 1 provided by Fred 13Ausubel and available from the Broad Bioimage Benchmark Collection [Ljosa et al., 14Nature Methods, 2012]." https://doi.org/10.1038/nmeth.2083 15""" 16 17import os 18import re 19from glob import glob 20from natsort import natsorted 21from typing import List, Optional, Tuple, Union 22 23import numpy as np 24import imageio.v3 as imageio 25from tqdm import tqdm 26from sklearn.model_selection import train_test_split 27 28from torch.utils.data import Dataset, DataLoader 29 30import torch_em 31 32from .. import util 33 34 35IMAGE_URL = "https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v2_images.zip" 36IMAGE_CHECKSUM = None 37 38GT_URL = "https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v1_foreground_eachworm.zip" 39GT_CHECKSUM = None 40 41WELL_PATTERN = re.compile(r"_([A-E]\d{2})_w(\d)_") 42 43 44def _get_well_and_channel(fname: str) -> Tuple[Optional[str], Optional[int]]: 45 """Extract the well id (e.g. 'A01') and channel number (1 or 2) from a raw image filename.""" 46 match = WELL_PATTERN.search(fname) 47 if match is None: 48 return None, None 49 return match.group(1), int(match.group(2)) 50 51 52def _merge_worm_masks(mask_paths: List[str]) -> np.ndarray: 53 """Merge per-worm binary masks into a single instance segmentation label image.""" 54 mask_paths = natsorted(mask_paths) 55 ref = imageio.imread(mask_paths[0]) 56 instances = np.zeros(ref.shape, dtype=np.int32) 57 for i, mask_path in enumerate(mask_paths, start=1): 58 mask = imageio.imread(mask_path) > 0 59 instances[mask] = i 60 return instances 61 62 63def _preprocess(data_dir: str, channel: int) -> str: 64 """Convert raw TIFs and per-worm ground truth PNGs to preprocessed H5 files.""" 65 import h5py 66 67 h5_dir = os.path.join(data_dir, f"h5_data_w{channel}") 68 if os.path.exists(h5_dir): 69 return h5_dir 70 os.makedirs(h5_dir, exist_ok=True) 71 72 raw_paths = glob(os.path.join(data_dir, "images", "*.tif")) 73 well_to_raw = {} 74 for raw_path in raw_paths: 75 well, this_channel = _get_well_and_channel(os.path.basename(raw_path)) 76 if well is None or this_channel != channel: 77 continue 78 well_to_raw[well] = raw_path 79 80 gt_dir = os.path.join(data_dir, "BBBC010_v1_foreground_eachworm") 81 for well, raw_path in tqdm(sorted(well_to_raw.items()), desc="Preprocessing BBBC010"): 82 mask_paths = glob(os.path.join(gt_dir, f"{well}_*_ground_truth.png")) 83 if len(mask_paths) == 0: 84 continue 85 86 raw = imageio.imread(raw_path) 87 instances = _merge_worm_masks(mask_paths) 88 89 h5_path = os.path.join(h5_dir, f"{well}.h5") 90 with h5py.File(h5_path, "w") as f: 91 f.create_dataset("raw", data=raw, compression="gzip") 92 f.create_dataset("labels", data=instances, compression="gzip") 93 94 return h5_dir 95 96 97def get_bbbc010_data(path: Union[os.PathLike, str], channel: int = 1, download: bool = False) -> str: 98 """Download and preprocess the BBBC010 dataset. 99 100 Args: 101 path: Filepath to a folder where the downloaded data will be saved. 102 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 103 Available channels: 1=brightfield, 2=GFP. 104 download: Whether to download the data if it is not present. 105 106 Returns: 107 The filepath to the preprocessed H5 data directory. 108 """ 109 data_dir = os.path.join(path, "BBBC010") 110 111 if not os.path.exists(data_dir): 112 os.makedirs(data_dir, exist_ok=True) 113 img_zip = os.path.join(path, "BBBC010_v2_images.zip") 114 gt_zip = os.path.join(path, "BBBC010_v1_foreground_eachworm.zip") 115 util.download_source(img_zip, IMAGE_URL, download, checksum=IMAGE_CHECKSUM) 116 util.download_source(gt_zip, GT_URL, download, checksum=GT_CHECKSUM) 117 util.unzip(img_zip, os.path.join(data_dir, "images")) 118 util.unzip(gt_zip, data_dir) 119 120 return _preprocess(data_dir, channel) 121 122 123def get_bbbc010_paths( 124 path: Union[os.PathLike, str], 125 split: Optional[str] = None, 126 channel: int = 1, 127 download: bool = False, 128) -> Tuple[List[str], List[str]]: 129 """Get paths to the BBBC010 data. 130 131 Args: 132 path: Filepath to a folder where the downloaded data will be saved. 133 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 134 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 135 Available channels: 1=brightfield, 2=GFP. 136 download: Whether to download the data if it is not present. 137 138 Returns: 139 List of filepaths for the image data (H5, key 'raw'). 140 List of filepaths for the label data (H5, key 'labels'). 141 """ 142 h5_dir = get_bbbc010_data(path, channel, download) 143 h5_paths = natsorted(glob(os.path.join(h5_dir, "*.h5"))) 144 145 if len(h5_paths) == 0: 146 raise RuntimeError(f"No preprocessed files found in {h5_dir}.") 147 148 if split is None: 149 return h5_paths, h5_paths 150 151 train_paths, test_paths = train_test_split(h5_paths, test_size=0.2, random_state=42) 152 train_paths, val_paths = train_test_split(train_paths, test_size=0.15, random_state=42) 153 154 split_map = {"train": train_paths, "val": val_paths, "test": test_paths} 155 assert split in split_map, f"'{split}' is not a valid split. Choose from {list(split_map)}." 156 selected = split_map[split] 157 return selected, selected 158 159 160def get_bbbc010_dataset( 161 path: Union[os.PathLike, str], 162 patch_shape: Tuple[int, int], 163 split: Optional[str] = None, 164 channel: int = 1, 165 download: bool = False, 166 **kwargs, 167) -> Dataset: 168 """Get the BBBC010 dataset for C. elegans instance segmentation. 169 170 Args: 171 path: Filepath to a folder where the downloaded data will be saved. 172 patch_shape: The patch shape to use for training. 173 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 174 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 175 Available channels: 1=brightfield, 2=GFP. 176 download: Whether to download the data if it is not present. 177 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 178 179 Returns: 180 The segmentation dataset. 181 """ 182 raw_paths, label_paths = get_bbbc010_paths(path, split, channel, download) 183 184 return torch_em.default_segmentation_dataset( 185 raw_paths=raw_paths, 186 raw_key="raw", 187 label_paths=label_paths, 188 label_key="labels", 189 patch_shape=patch_shape, 190 **kwargs, 191 ) 192 193 194def get_bbbc010_loader( 195 path: Union[os.PathLike, str], 196 batch_size: int, 197 patch_shape: Tuple[int, int], 198 split: Optional[str] = None, 199 channel: int = 1, 200 download: bool = False, 201 **kwargs, 202) -> DataLoader: 203 """Get the BBBC010 dataloader for C. elegans instance segmentation. 204 205 Args: 206 path: Filepath to a folder where the downloaded data will be saved. 207 batch_size: The batch size for training. 208 patch_shape: The patch shape to use for training. 209 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 210 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 211 Available channels: 1=brightfield, 2=GFP. 212 download: Whether to download the data if it is not present. 213 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 214 215 Returns: 216 The DataLoader. 217 """ 218 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 219 dataset = get_bbbc010_dataset(path, patch_shape, split, channel, download, **ds_kwargs) 220 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
98def get_bbbc010_data(path: Union[os.PathLike, str], channel: int = 1, download: bool = False) -> str: 99 """Download and preprocess the BBBC010 dataset. 100 101 Args: 102 path: Filepath to a folder where the downloaded data will be saved. 103 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 104 Available channels: 1=brightfield, 2=GFP. 105 download: Whether to download the data if it is not present. 106 107 Returns: 108 The filepath to the preprocessed H5 data directory. 109 """ 110 data_dir = os.path.join(path, "BBBC010") 111 112 if not os.path.exists(data_dir): 113 os.makedirs(data_dir, exist_ok=True) 114 img_zip = os.path.join(path, "BBBC010_v2_images.zip") 115 gt_zip = os.path.join(path, "BBBC010_v1_foreground_eachworm.zip") 116 util.download_source(img_zip, IMAGE_URL, download, checksum=IMAGE_CHECKSUM) 117 util.download_source(gt_zip, GT_URL, download, checksum=GT_CHECKSUM) 118 util.unzip(img_zip, os.path.join(data_dir, "images")) 119 util.unzip(gt_zip, data_dir) 120 121 return _preprocess(data_dir, channel)
Download and preprocess the BBBC010 dataset.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
- download: Whether to download the data if it is not present.
Returns:
The filepath to the preprocessed H5 data directory.
124def get_bbbc010_paths( 125 path: Union[os.PathLike, str], 126 split: Optional[str] = None, 127 channel: int = 1, 128 download: bool = False, 129) -> Tuple[List[str], List[str]]: 130 """Get paths to the BBBC010 data. 131 132 Args: 133 path: Filepath to a folder where the downloaded data will be saved. 134 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 135 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 136 Available channels: 1=brightfield, 2=GFP. 137 download: Whether to download the data if it is not present. 138 139 Returns: 140 List of filepaths for the image data (H5, key 'raw'). 141 List of filepaths for the label data (H5, key 'labels'). 142 """ 143 h5_dir = get_bbbc010_data(path, channel, download) 144 h5_paths = natsorted(glob(os.path.join(h5_dir, "*.h5"))) 145 146 if len(h5_paths) == 0: 147 raise RuntimeError(f"No preprocessed files found in {h5_dir}.") 148 149 if split is None: 150 return h5_paths, h5_paths 151 152 train_paths, test_paths = train_test_split(h5_paths, test_size=0.2, random_state=42) 153 train_paths, val_paths = train_test_split(train_paths, test_size=0.15, random_state=42) 154 155 split_map = {"train": train_paths, "val": val_paths, "test": test_paths} 156 assert split in split_map, f"'{split}' is not a valid split. Choose from {list(split_map)}." 157 selected = split_map[split] 158 return selected, selected
Get paths to the BBBC010 data.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- split: The data split to use. One of 'train', 'val', 'test', or None (use all).
- channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data (H5, key 'raw'). List of filepaths for the label data (H5, key 'labels').
161def get_bbbc010_dataset( 162 path: Union[os.PathLike, str], 163 patch_shape: Tuple[int, int], 164 split: Optional[str] = None, 165 channel: int = 1, 166 download: bool = False, 167 **kwargs, 168) -> Dataset: 169 """Get the BBBC010 dataset for C. elegans instance segmentation. 170 171 Args: 172 path: Filepath to a folder where the downloaded data will be saved. 173 patch_shape: The patch shape to use for training. 174 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 175 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 176 Available channels: 1=brightfield, 2=GFP. 177 download: Whether to download the data if it is not present. 178 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 179 180 Returns: 181 The segmentation dataset. 182 """ 183 raw_paths, label_paths = get_bbbc010_paths(path, split, channel, download) 184 185 return torch_em.default_segmentation_dataset( 186 raw_paths=raw_paths, 187 raw_key="raw", 188 label_paths=label_paths, 189 label_key="labels", 190 patch_shape=patch_shape, 191 **kwargs, 192 )
Get the BBBC010 dataset for C. elegans instance segmentation.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- patch_shape: The patch shape to use for training.
- split: The data split to use. One of 'train', 'val', 'test', or None (use all).
- channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
195def get_bbbc010_loader( 196 path: Union[os.PathLike, str], 197 batch_size: int, 198 patch_shape: Tuple[int, int], 199 split: Optional[str] = None, 200 channel: int = 1, 201 download: bool = False, 202 **kwargs, 203) -> DataLoader: 204 """Get the BBBC010 dataloader for C. elegans instance segmentation. 205 206 Args: 207 path: Filepath to a folder where the downloaded data will be saved. 208 batch_size: The batch size for training. 209 patch_shape: The patch shape to use for training. 210 split: The data split to use. One of 'train', 'val', 'test', or None (use all). 211 channel: The imaging channel to use as raw input. Default: 1 (brightfield). 212 Available channels: 1=brightfield, 2=GFP. 213 download: Whether to download the data if it is not present. 214 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 215 216 Returns: 217 The DataLoader. 218 """ 219 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 220 dataset = get_bbbc010_dataset(path, patch_shape, split, channel, download, **ds_kwargs) 221 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BBBC010 dataloader for C. elegans instance segmentation.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The data split to use. One of 'train', 'val', 'test', or None (use all).
- channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.