Team Ai
Datasetpublic

angelolab/ark_example

This dataset contains 11 Field of Views (FOVs), each with 22 channels.

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes2.8kdownloads
ark_example.py232 linesDownload Raw Back to root
1# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.2#3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License at6#7#     http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14 15"""16This dataset contains example data for running through the multiplexed imaging data pipeline in17Ark Analysis: https://github.com/angelolab/ark-analysis.18 19 20Dataset Fov renaming:21 22TMA2_R8C3 -> fov023TMA6_R4C5 -> fov124TMA7_R5C4 -> fov225TMA10_R7C3 -> fov326TMA11_R9C6 -> fov427TMA13_R8C5 -> fov528TMA17_R9C2 -> fov629TMA18_R9C2 -> fov730TMA21_R2C5 -> fov831TMA21_R12C6 -> fov932TMA24_R9C1 -> fov1033 34"""35 36import datasets37import pathlib38 39# Find for instance the citation on arxiv or on the dataset repo/website40_CITATION = """\41@InProceedings{huggingface:dataset,42title = {Ark Analysis Example Dataset},43author={Angelo Lab},44year={2022}45}46"""47 48# TODO: Add description of the dataset here49# You can copy an official description50_DESCRIPTION = """\51This dataset contains 11 Field of Views (FOVs), each with 22 channels. 52"""53 54_HOMEPAGE = "https://github.com/angelolab/ark-analysis"55 56_LICENSE = "https://github.com/angelolab/ark-analysis/blob/main/LICENSE"57 58# The HuggingFace Datasets library doesn't host the datasets but only points to the original files.59# This can be an arbitrary nested dict/list of URLs (see below in `_split_generators` method)60 61_URL_DATA = {62    "image_data": "data/image_data.zip",63    "cell_table": "data/segmentation/cell_table.zip",64    "deepcell_output": "data/segmentation/deepcell_output.zip",65    "example_pixel_output_dir": "data/pixie/example_pixel_output_dir.zip",66    "example_cell_output_dir": "data/pixie/example_cell_output_dir.zip",67    "spatial_lda": "data/spatial_analysis/spatial_lda.zip",68    "post_clustering": "data/post_clustering.zip",69    "ome_tiff": "data/ome_tiff.zip",70    "ez_seg_data": "data/ez_seg_data.zip"71}72 73_URL_DATASET_CONFIGS = {74    "segment_image_data": {"image_data": _URL_DATA["image_data"]},75    "cluster_pixels": {76        "image_data": _URL_DATA["image_data"],77        "cell_table": _URL_DATA["cell_table"],78        "deepcell_output": _URL_DATA["deepcell_output"],79    },80    "cluster_cells": {81        "image_data": _URL_DATA["image_data"],82        "cell_table": _URL_DATA["cell_table"],83        "deepcell_output": _URL_DATA["deepcell_output"],84        "example_pixel_output_dir": _URL_DATA["example_pixel_output_dir"],85    },86    "post_clustering": {87        "image_data": _URL_DATA["image_data"],88        "cell_table": _URL_DATA["cell_table"],89        "deepcell_output": _URL_DATA["deepcell_output"],90        "example_cell_output_dir": _URL_DATA["example_cell_output_dir"],91    },92    "fiber_segmentation": {93        "image_data": _URL_DATA["image_data"],94    },95    "LDA_preprocessing": {96        "image_data": _URL_DATA["image_data"],97        "cell_table": _URL_DATA["cell_table"],98    },99    "LDA_training_inference": {100        "image_data": _URL_DATA["image_data"],101        "cell_table": _URL_DATA["cell_table"],102        "spatial_lda": _URL_DATA["spatial_lda"],103    },104    "neighborhood_analysis": {105        "image_data": _URL_DATA["image_data"],106        "cell_table": _URL_DATA["cell_table"],107        "deepcell_output": _URL_DATA["deepcell_output"],108    },109    "pairwise_spatial_enrichment": {110        "image_data": _URL_DATA["image_data"],111        "cell_table": _URL_DATA["cell_table"],112        "deepcell_output": _URL_DATA["deepcell_output"],113        "post_clustering": _URL_DATA["post_clustering"],114    },115    "ome_tiff": {116        "ome_tiff": _URL_DATA["ome_tiff"],117    },118    "ez_seg_data": {119        "ez_seg_data": _URL_DATA["ez_seg_data"]120    }121}122 123 124# Note: Name of the dataset usually match the script name with CamelCase instead of snake_case125class ArkExample(datasets.GeneratorBasedBuilder):126    """The Dataset consists of 11 FOVs"""127 128    VERSION = datasets.Version("0.0.5")129 130    # You will be able to load one or the other configurations in the following list with131    BUILDER_CONFIGS = [132        datasets.BuilderConfig(133            name="segment_image_data",134            version=VERSION,135            description="This configuration contains data used by notebook 1 - Segment Image Data.",136        ),137        datasets.BuilderConfig(138            name="cluster_pixels",139            version=VERSION,140            description="This configuration contains data used by notebook 2 - Pixel Clustering (Pixie Pipeline #1).",141        ),142        datasets.BuilderConfig(143            name="cluster_cells",144            version=VERSION,145            description="This configuration contains data used by notebook 3 - Cell Clustering (Pixie Pipeline #2).",146        ),147        datasets.BuilderConfig(148            name="post_clustering",149            version=VERSION,150            description="This configuration contains data used by notebook 4 - Post Clustering.",151        ),152        datasets.BuilderConfig(153            name="fiber_segmentation",154            version=VERSION,155            description="This configuration contains data used by the Fiber Segmentation Notebook.",156        ),157        datasets.BuilderConfig(158            name="LDA_preprocessing",159            version=VERSION,160            description="This configuration contains data used by the Spatial LDA - Preprocessing Notebook."161        ),162        datasets.BuilderConfig(163            name="LDA_training_inference",164            version=VERSION,165            description="This configuration contains data used by the Spatial LDA - Training and Inference Notebook."166        ),167        datasets.BuilderConfig(168            name="neighborhood_analysis",169            version=VERSION,170            description="This configuration contains data used by the Neighborhood Analysis Notebook."171        ),172        datasets.BuilderConfig(173            name="pairwise_spatial_enrichment",174            version=VERSION,175            description="This configuration contains data used by the Pairwise Spatial Enrichment Notebook."176        ),177        datasets.BuilderConfig(178            name="ome_tiff",179            version=VERSION,180            description="This configuration contains an OME-TIFF format of FOV1. Intended to be used with the OME-TIFF Conversion Notebook."181        ),182        datasets.BuilderConfig(183            name="ez_seg_data",184            version=VERSION,185            description="This configuration contains the data used by the ezSegmenter notebook."186        )187    ]188 189    def _info(self):190        # This is the name of the configuration selected in BUILDER_CONFIGS above191        if self.config.name in list(_URL_DATASET_CONFIGS.keys()):192            features = datasets.Features(193                {f: datasets.Value("string") for f in _URL_DATASET_CONFIGS[self.config.name].keys()}194            )195        else:196            ValueError(f"Dataset name is incorrect, options include {list(_URL_DATASET_CONFIGS.keys())}")197        return datasets.DatasetInfo(198            # This is the description that will appear on the datasets page.199            description=_DESCRIPTION,200            # This defines the different columns of the dataset and their types201            features=features,  # Here we define them above because they are different between the two configurations202            # If there's a common (input, target) tuple from the features, uncomment supervised_keys line below and203            # specify them. They'll be used if as_supervised=True in builder.as_dataset.204            # supervised_keys=("sentence", "label"),205            # Homepage of the dataset for documentation206            homepage=_HOMEPAGE,207            # License for the dataset if available208            license=_LICENSE,209            # Citation for the dataset210            citation=_CITATION,211        )212 213    def _split_generators(self, dl_manager):214        # This method is tasked with downloading/extracting the data and defining the splits depending on the configuration215        urls = _URL_DATASET_CONFIGS[self.config.name]216        data_dirs = {}217        for data_name, url in urls.items():218            dl_path = pathlib.Path(dl_manager.download_and_extract(url))219            data_dirs[data_name] = dl_path220 221        return [222            datasets.SplitGenerator(223                name=self.config.name,224                # These kwargs will be passed to _generate_examples225                gen_kwargs={"dataset_paths": data_dirs},226            ),227        ]228 229    # method parameters are unpacked from `gen_kwargs` as given in `_split_generators`230    def _generate_examples(self, dataset_paths):231        yield self.config.name, dataset_paths232