Team Ai
Datasetpublic

asahi417/multi-domain-document-classification

multi_domain_document_classification Multi-domain document classification datasets. Biomedical: chemprot, rct-sample Computer Science: citation_intent, sciie Customer Review: amcd, yelp_review Social Media: tweet_eval_irony, tweet_eval_hate, tweet_eval_emotion The yelp_review dataset is randomly downsampled to 2000/2000/8000 for test/validation/train. chemprot citation_intent hyperpartisan_news rct_sample sciie amcd yelp_review tweet_eval_irony tweet_eval_hate… See the full description on the dataset page: https://huggingface.co/datasets/asahi417/multi-domain-document-classification.

sourceHugging Faceupdated 4y agoView on Hugging Face
0likes230downloads
multi_domain_document_classification.py100 linesDownload Raw Back to root
1"""Multi domain document classification dataset used in [https://arxiv.org/pdf/2004.10964.pdf](https://arxiv.org/pdf/2004.10964.pdf)"""2import json3from itertools import chain4import datasets5 6logger = datasets.logging.get_logger(__name__)7_DESCRIPTION = """Multi domain document classification dataset used in [https://arxiv.org/pdf/2004.10964.pdf](https://arxiv.org/pdf/2004.10964.pdf)"""8_NAME = "multi_domain_document_classification"9_VERSION = "0.2.3"10_CITATION = """11@inproceedings{dontstoppretraining2020,12 author = {Suchin Gururangan and Ana Marasović and Swabha Swayamdipta and Kyle Lo and Iz Beltagy and Doug Downey and Noah A. Smith},13 title = {Don't Stop Pretraining: Adapt Language Models to Domains and Tasks},14 year = {2020},15 booktitle = {Proceedings of ACL},16}17"""18 19_HOME_PAGE = "https://github.com/asahi417/m3"20_URL = f'https://huggingface.co/datasets/asahi417/{_NAME}/raw/main/dataset'21_DATA_TYPE = ["chemprot", "citation_intent", "hyperpartisan_news", "rct_sample", "sciie", "amcd",22              "yelp_review", "tweet_eval_irony", "tweet_eval_hate", "tweet_eval_emotion"]23_URLS = {24    k:25        {26            str(datasets.Split.TEST): [f'{_URL}/{k}/test.jsonl'],27            str(datasets.Split.TRAIN): [f'{_URL}/{k}/train.jsonl'],28            str(datasets.Split.VALIDATION): [f'{_URL}/{k}/dev.jsonl']29        }30    for k in _DATA_TYPE31}32_LABELS = {33    "chemprot": {"ACTIVATOR": 0, "AGONIST": 1, "AGONIST-ACTIVATOR": 2, "AGONIST-INHIBITOR": 3, "ANTAGONIST": 4, "DOWNREGULATOR": 5, "INDIRECT-DOWNREGULATOR": 6, "INDIRECT-UPREGULATOR": 7, "INHIBITOR": 8, "PRODUCT-OF": 9, "SUBSTRATE": 10, "SUBSTRATE_PRODUCT-OF": 11, "UPREGULATOR": 12},34    "citation_intent": {"Background": 0, "CompareOrContrast": 1, "Extends": 2, "Future": 3, "Motivation": 4, "Uses": 5},35    "hyperpartisan_news": {"false": 0, "true": 1},36    "rct_sample": {"BACKGROUND": 0, "CONCLUSIONS": 1, "METHODS": 2, "OBJECTIVE": 3, "RESULTS": 4},37    "sciie": {"COMPARE": 0, "CONJUNCTION": 1, "EVALUATE-FOR": 2, "FEATURE-OF": 3, "HYPONYM-OF": 4, "PART-OF": 5, "USED-FOR": 6},38    "amcd": {"false": 0, "true": 1},39    "yelp_review": {"5 star": 4, "4 star": 3, "3 star": 2, "2 star": 1, "1 star": 0},40    "tweet_eval_irony": {"non_irony":0, "irony": 1},41    "tweet_eval_hate": {"non_hate": 0, "hate": 1},42    "tweet_eval_emotion": {"anger": 0, "joy": 1, "optimism": 2, "sadness": 3}43}44 45 46class MultiDomainDocumentClassificationConfig(datasets.BuilderConfig):47    """BuilderConfig"""48 49    def __init__(self, **kwargs):50        """BuilderConfig.51 52        Args:53          **kwargs: keyword arguments forwarded to super.54        """55        super(MultiDomainDocumentClassificationConfig, self).__init__(**kwargs)56 57 58class MultiDomainDocumentClassification(datasets.GeneratorBasedBuilder):59    """Dataset."""60 61    BUILDER_CONFIGS = [62        MultiDomainDocumentClassificationConfig(63            name=k, version=datasets.Version(_VERSION), description=_DESCRIPTION64        ) for k in _DATA_TYPE65    ]66 67    def _split_generators(self, dl_manager):68        downloaded_file = dl_manager.download_and_extract(_URLS[self.config.name])69        return [70            datasets.SplitGenerator(name=i, gen_kwargs={"filepaths": downloaded_file[str(i)]})71            for i in [datasets.Split.TRAIN, datasets.Split.VALIDATION, datasets.Split.TEST]72        ]73 74    def _generate_examples(self, filepaths):75        _key = 076        for filepath in filepaths:77            logger.info(f"generating examples from = {filepath}")78            with open(filepath, encoding="utf-8") as f:79                _list = [i for i in f.read().split('\n') if len(i) > 0]80                for i in _list:81                    data = json.loads(i)82                    yield _key, data83                    _key += 184 85    def _info(self):86        label2id = sorted(_LABELS[self.config.name].items(), key=lambda x: x[1])87        label = [i[0] for i in label2id]88        return datasets.DatasetInfo(89            description=_DESCRIPTION,90            features=datasets.Features(91                {92                    "text": datasets.Value("string"),93                    "label": datasets.features.ClassLabel(names=label),94                }95            ),96            supervised_keys=None,97            homepage=_HOME_PAGE,98            citation=_CITATION,99        )100