SEACrowd/mtop_intent_classification
This dataset contains annotated utterances from 6 languages, including Thai, for semantic parsing. Queries corresponding to the chosen domains are crowdsourced. Two subsets are included in this dataset: 'domain' (eg. 'news', 'people', 'weather') and 'intent' (eg. 'GET_MESSAGE', 'STOP_MUSIC', 'END_CALL')
086
1# coding=utf-82# Copyright 2022 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15from typing import Dict, List, Tuple16 17import datasets18 19from seacrowd.sea_datasets.mtop_intent_classification.labels import (20 DOMAIN_LABELS, INTENT_LABELS)21from seacrowd.utils import schemas22from seacrowd.utils.configs import SEACrowdConfig23from seacrowd.utils.constants import Licenses, Tasks24 25_CITATION = """\26@inproceedings{li-etal-2021-mtop,27 author = {Li, Haoran and Arora, Abhinav and Chen, Shuochi and Gupta, Anchit and Gupta, Sonal and Mehdad, Yashar},28 title = {MTOP: A Comprehensive Multilingual Task-Oriented Semantic Parsing Benchmark},29 booktitle = {Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume},30 publisher = {Association for Computational Linguistics},31 year = {2021},32 url = {https://aclanthology.org/2021.eacl-main.257},33 doi = {10.18653/v1/2021.eacl-main.257},34 pages = {2950-2962},35}36"""37_LOCAL = False38_LANGUAGES = ["tha"]39_DATASETNAME = "mtop_intent_classification"40_DESCRIPTION = """41This dataset contains annotated utterances from 6 languages, including Thai,42for semantic parsing. Queries corresponding to the chosen domains are crowdsourced.43 Two subsets are included in this dataset: 'domain' (eg. 'news', 'people', 'weather')44 and 'intent' (eg. 'GET_MESSAGE', 'STOP_MUSIC', 'END_CALL')45"""46 47_HOMEPAGE = "https://huggingface.co/mteb"48_LICENSE = Licenses.CC_BY_SA_4_0.value # Found in original dataset (not HF) linked in paper49_URL = "https://huggingface.co/datasets/mteb/"50 51 52_SUPPORTED_TASKS = [Tasks.INTENT_CLASSIFICATION]53_SOURCE_VERSION = "1.0.1"54_SEACROWD_VERSION = "2025.04.22"55 56 57class MTOPIntentClassificationDataset(datasets.GeneratorBasedBuilder):58 """Dataset of Thai sentences and their domains or intents."""59 60 SOURCE_VERSION = datasets.Version(_SOURCE_VERSION)61 SEACROWD_VERSION = datasets.Version(_SEACROWD_VERSION)62 SUBSETS = ["domain", "intent"]63 64 BUILDER_CONFIGS = [65 SEACrowdConfig(66 name=f"{_DATASETNAME}_{subset}_source",67 version=datasets.Version(_SOURCE_VERSION),68 description=f"{_DATASETNAME} source schema for {subset} subset",69 schema="source",70 subset_id=f"{_DATASETNAME}_{subset}",71 )72 for subset in SUBSETS73 ] + [74 SEACrowdConfig(75 name=f"{_DATASETNAME}_{subset}_seacrowd_text",76 version=datasets.Version(_SEACROWD_VERSION),77 description=f"{_DATASETNAME} SEACrowd schema for {subset} subset",78 schema="seacrowd_text",79 subset_id=f"{_DATASETNAME}_{subset}",80 )81 for subset in SUBSETS82 ]83 84 DEFAULT_CONFIG_NAME = f"{_DATASETNAME}_domain_source"85 86 def _info(self) -> datasets.DatasetInfo:87 if self.config.schema == "source":88 features = datasets.Features(89 {90 "id": datasets.Value("int64"),91 "text": datasets.Value("string"),92 "label": datasets.Value("int32"),93 "label_text": datasets.Value("string"),94 }95 )96 97 elif self.config.schema == "seacrowd_text":98 if self.config.subset_id.endswith("domain"):99 labels = DOMAIN_LABELS100 elif self.config.subset_id.endswith("intent"):101 labels = INTENT_LABELS102 else:103 raise ValueError(f"Received unexpected schema name {self.config.name}")104 features = schemas.text_features(label_names=labels)105 106 return datasets.DatasetInfo(107 description=_DESCRIPTION,108 features=features,109 homepage=_HOMEPAGE,110 license=_LICENSE,111 citation=_CITATION,112 )113 114 def _split_generators(self, dl_manager: datasets.DownloadManager) -> List[datasets.SplitGenerator]:115 # dl_manager not used since dataloader uses HF `load_dataset`116 return [datasets.SplitGenerator(name=split, gen_kwargs={"split": split._name}) for split in (datasets.Split.TRAIN, datasets.Split.VALIDATION, datasets.Split.TEST)]117 118 def _load_hf_data_from_remote(self, split: str) -> datasets.DatasetDict:119 """Load dataset from HuggingFace."""120 if self.config.subset_id.endswith("domain"):121 subset = "domain"122 elif self.config.subset_id.endswith("intent"):123 subset = "intent"124 else:125 raise ValueError(f"Received unexpected schema name {self.config.name}")126 HF_REMOTE_REF = f"mteb/mtop_{subset}"127 _hf_dataset_source = datasets.load_dataset(HF_REMOTE_REF, "th", split=split)128 return _hf_dataset_source129 130 def _generate_examples(self, split: str) -> Tuple[int, Dict]:131 """Yields examples as (key, example) tuples."""132 data = self._load_hf_data_from_remote(split=split)133 for index, row in enumerate(data):134 if self.config.schema == "source":135 example = row136 137 elif self.config.schema == "seacrowd_text":138 example = {"id": str(index), "text": row["text"], "label": row["label_text"]}139 yield index, example140 