Team Ai
Datasetpublic

SEACrowd/mtop_intent_classification

This dataset contains annotated utterances from 6 languages, including Thai, for semantic parsing. Queries corresponding to the chosen domains are crowdsourced. Two subsets are included in this dataset: 'domain' (eg. 'news', 'people', 'weather') and 'intent' (eg. 'GET_MESSAGE', 'STOP_MUSIC', 'END_CALL')

sourceHugging Facecc-by-sa-4.0updated 1y agoView on Hugging Face
0likes86downloads
mtop_intent_classification.py140 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2022 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15from typing import Dict, List, Tuple16 17import datasets18 19from seacrowd.sea_datasets.mtop_intent_classification.labels import (20    DOMAIN_LABELS, INTENT_LABELS)21from seacrowd.utils import schemas22from seacrowd.utils.configs import SEACrowdConfig23from seacrowd.utils.constants import Licenses, Tasks24 25_CITATION = """\26@inproceedings{li-etal-2021-mtop,27  author    = {Li, Haoran and Arora, Abhinav and Chen, Shuochi and Gupta, Anchit and Gupta, Sonal and Mehdad, Yashar},28  title     = {MTOP: A Comprehensive Multilingual Task-Oriented Semantic Parsing Benchmark},29  booktitle   = {Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume},30  publisher = {Association for Computational Linguistics},31  year      = {2021},32  url       = {https://aclanthology.org/2021.eacl-main.257},33  doi       = {10.18653/v1/2021.eacl-main.257},34  pages    = {2950-2962},35}36"""37_LOCAL = False38_LANGUAGES = ["tha"]39_DATASETNAME = "mtop_intent_classification"40_DESCRIPTION = """41This dataset contains annotated utterances from 6 languages, including Thai,42for semantic parsing. Queries corresponding to the chosen domains are crowdsourced.43 Two subsets are included in this dataset: 'domain' (eg. 'news', 'people', 'weather')44 and 'intent' (eg. 'GET_MESSAGE', 'STOP_MUSIC', 'END_CALL')45"""46 47_HOMEPAGE = "https://huggingface.co/mteb"48_LICENSE = Licenses.CC_BY_SA_4_0.value  # Found in original dataset (not HF) linked in paper49_URL = "https://huggingface.co/datasets/mteb/"50 51 52_SUPPORTED_TASKS = [Tasks.INTENT_CLASSIFICATION]53_SOURCE_VERSION = "1.0.1"54_SEACROWD_VERSION = "2025.04.22"55 56 57class MTOPIntentClassificationDataset(datasets.GeneratorBasedBuilder):58    """Dataset of Thai sentences and their domains or intents."""59 60    SOURCE_VERSION = datasets.Version(_SOURCE_VERSION)61    SEACROWD_VERSION = datasets.Version(_SEACROWD_VERSION)62    SUBSETS = ["domain", "intent"]63 64    BUILDER_CONFIGS = [65        SEACrowdConfig(66            name=f"{_DATASETNAME}_{subset}_source",67            version=datasets.Version(_SOURCE_VERSION),68            description=f"{_DATASETNAME} source schema for {subset} subset",69            schema="source",70            subset_id=f"{_DATASETNAME}_{subset}",71        )72        for subset in SUBSETS73    ] + [74        SEACrowdConfig(75            name=f"{_DATASETNAME}_{subset}_seacrowd_text",76            version=datasets.Version(_SEACROWD_VERSION),77            description=f"{_DATASETNAME} SEACrowd schema for {subset} subset",78            schema="seacrowd_text",79            subset_id=f"{_DATASETNAME}_{subset}",80        )81        for subset in SUBSETS82    ]83 84    DEFAULT_CONFIG_NAME = f"{_DATASETNAME}_domain_source"85 86    def _info(self) -> datasets.DatasetInfo:87        if self.config.schema == "source":88            features = datasets.Features(89                {90                    "id": datasets.Value("int64"),91                    "text": datasets.Value("string"),92                    "label": datasets.Value("int32"),93                    "label_text": datasets.Value("string"),94                }95            )96 97        elif self.config.schema == "seacrowd_text":98            if self.config.subset_id.endswith("domain"):99                labels = DOMAIN_LABELS100            elif self.config.subset_id.endswith("intent"):101                labels = INTENT_LABELS102            else:103                raise ValueError(f"Received unexpected schema name {self.config.name}")104            features = schemas.text_features(label_names=labels)105 106        return datasets.DatasetInfo(107            description=_DESCRIPTION,108            features=features,109            homepage=_HOMEPAGE,110            license=_LICENSE,111            citation=_CITATION,112        )113 114    def _split_generators(self, dl_manager: datasets.DownloadManager) -> List[datasets.SplitGenerator]:115        # dl_manager not used since dataloader uses HF `load_dataset`116        return [datasets.SplitGenerator(name=split, gen_kwargs={"split": split._name}) for split in (datasets.Split.TRAIN, datasets.Split.VALIDATION, datasets.Split.TEST)]117 118    def _load_hf_data_from_remote(self, split: str) -> datasets.DatasetDict:119        """Load dataset from HuggingFace."""120        if self.config.subset_id.endswith("domain"):121            subset = "domain"122        elif self.config.subset_id.endswith("intent"):123            subset = "intent"124        else:125            raise ValueError(f"Received unexpected schema name {self.config.name}")126        HF_REMOTE_REF = f"mteb/mtop_{subset}"127        _hf_dataset_source = datasets.load_dataset(HF_REMOTE_REF, "th", split=split)128        return _hf_dataset_source129 130    def _generate_examples(self, split: str) -> Tuple[int, Dict]:131        """Yields examples as (key, example) tuples."""132        data = self._load_hf_data_from_remote(split=split)133        for index, row in enumerate(data):134            if self.config.schema == "source":135                example = row136 137            elif self.config.schema == "seacrowd_text":138                example = {"id": str(index), "text": row["text"], "label": row["label_text"]}139            yield index, example140