Team Ai
Datasetpublic

jglaser/binding_affinity

A dataset to fine-tune language models on protein-ligand binding affinity prediction.

sourceHugging Faceupdated 5y agoView on Hugging Face
21likes1.8kdownloads
binding_affinity.py148 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15"""TODO: A dataset of protein sequences, ligand SMILES and binding affinities."""16 17import huggingface_hub18import os19import pyarrow.parquet as pq20import datasets21 22 23# TODO: Add BibTeX citation24# Find for instance the citation on arxiv or on the dataset repo/website25_CITATION = """\26@InProceedings{huggingface:dataset,27title = {jglaser/binding_affinity},28author={Jens Glaser, ORNL29},30year={2021}31}32"""33 34# TODO: Add description of the dataset here35# You can copy an official description36_DESCRIPTION = """\37A dataset to fine-tune language models on protein-ligand binding affinity prediction.38"""39 40# TODO: Add a link to an official homepage for the dataset here41_HOMEPAGE = ""42 43# TODO: Add the licence for the dataset here if you can find it44_LICENSE = "BSD two-clause"45 46# TODO: Add link to the official dataset URLs here47# The HuggingFace dataset library don't host the datasets but only point to the original files48# This can be an arbitrary nested dict/list of URLs (see below in `_split_generators` method)49_URL = "https://huggingface.co/datasets/jglaser/binding_affinity/resolve/main/"50_data_dir = "data/"51_file_names = {'default': _data_dir+'all.parquet',52               'no_kras': _data_dir+'all_nokras.parquet',53               'cov': _data_dir+'cov.parquet'}54 55_URLs = {name: _URL+_file_names[name] for name in _file_names}56 57 58# TODO: Name of the dataset usually match the script name with CamelCase instead of snake_case59class BindingAffinity(datasets.ArrowBasedBuilder):60    """List of protein sequences, ligand SMILES and binding affinities."""61 62    VERSION = datasets.Version("1.4.1")63 64    def _info(self):65        # TODO: This method specifies the datasets.DatasetInfo object which contains informations and typings for the dataset66        #if self.config.name == "first_domain":  # This is the name of the configuration selected in BUILDER_CONFIGS above67        #    features = datasets.Features(68        #        {69        #            "sentence": datasets.Value("string"),70        #             "option1": datasets.Value("string"),71        #            "answer": datasets.Value("string")72        #            # These are the features of your dataset like images, labels ...73        #        }74        #    )75        #else:  # This is an example to show how to have different features for "first_domain" and "second_domain"76        features = datasets.Features(77            {78                "seq": datasets.Value("string"),79                "smiles": datasets.Value("string"),80                "affinity_uM": datasets.Value("float"),81                "neg_log10_affinity_M": datasets.Value("float"),82                "smiles_can": datasets.Value("string"),83                "affinity": datasets.Value("float"),84                # These are the features of your dataset like images, labels ...85            }86        )87        return datasets.DatasetInfo(88            # This is the description that will appear on the datasets page.89            description=_DESCRIPTION,90            # This defines the different columns of the dataset and their types91            features=features,  # Here we define them above because they are different between the two configurations92            # If there's a common (input, target) tuple from the features,93            # specify them here. They'll be used if as_supervised=True in94            # builder.as_dataset.95            supervised_keys=None,96            # Homepage of the dataset for documentation97            homepage=_HOMEPAGE,98            # License for the dataset if available99            license=_LICENSE,100            # Citation for the dataset101            citation=_CITATION,102        )103 104    def _split_generators(self, dl_manager):105        """Returns SplitGenerators."""106        # TODO: This method is tasked with downloading/extracting the data and defining the splits depending on the configuration107        # If several configurations are possible (listed in BUILDER_CONFIGS), the configuration selected by the user is in self.config.name108 109        # dl_manager is a datasets.download.DownloadManager that can be used to download and extract URLs110        # It can accept any type or nested list/dict and will give back the same structure with the url replaced with path to local files.111        # By default the archives will be extracted and a path to a cached folder where they are extracted is returned instead of the archive112        files = dl_manager.download_and_extract(_URLs)113 114        return [115            datasets.SplitGenerator(116                # These kwargs will be passed to _generate_examples117                name=datasets.Split.TRAIN,118                gen_kwargs={119                    'filepath': files["default"],120                },121            ),122 123           datasets.SplitGenerator(124                name='no_kras',125                # These kwargs will be passed to _generate_examples126                gen_kwargs={127                    "filepath": files["no_kras"],128                },129            ),130 131           datasets.SplitGenerator(132                name='covalent',133                # These kwargs will be passed to _generate_examples134                gen_kwargs={135                    "filepath": files["cov"],136                },137            ),138        ]139 140    def _generate_tables(141        self, filepath142    ):143        from pyarrow import fs144        local = fs.LocalFileSystem()145 146        for i, f in enumerate([filepath]):147            yield i, pq.read_table(f,filesystem=local)148