Team Ai
Datasetpublic

google-research-datasets/taskmaster2

Taskmaster is dataset for goal oriented conversations. The Taskmaster-2 dataset consists of 17,289 dialogs in the seven domains which include restaurants, food ordering, movies, hotels, flights, music and sports. Unlike Taskmaster-1, which includes both written "self-dialogs" and spoken two-person dialogs, Taskmaster-2 consists entirely of spoken two-person dialogs. In addition, while Taskmaster-1 is almost exclusively task-based, Taskmaster-2 contains a good number of search- and recommendation-oriented dialogs. All dialogs in this release were created using a Wizard of Oz (WOz) methodology in which crowdsourced workers played the role of a 'user' and trained call center operators played the role of the 'assistant'. In this way, users were led to believe they were interacting with an automated system that “spoke” using text-to-speech (TTS) even though it was in fact a human behind the scenes. As a result, users could express themselves however they chose in the context of an automated interface.

sourceHugging Facecc-by-4.0updated 3y agoView on Hugging Face
7likes2.5kdownloads
taskmaster2.py128 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15"""Taskmaster: A dataset for goal oriented conversations."""16 17 18import json19 20import datasets21 22 23_CITATION = """\24@inproceedings{48484,25title	= {Taskmaster-1: Toward a Realistic and Diverse Dialog Dataset},26author	= {Bill Byrne and Karthik Krishnamoorthi and Chinnadhurai Sankar and Arvind Neelakantan and Daniel Duckworth and Semih Yavuz and Ben Goodrich and Amit Dubey and Kyu-Young Kim and Andy Cedilnik},27year	= {2019}28}29"""30 31_DESCRIPTION = """\32Taskmaster is dataset for goal oriented conversations. The Taskmaster-2 dataset consists of 17,289 dialogs \33in the seven domains which include restaurants, food ordering, movies, hotels, flights, music and sports. \34Unlike Taskmaster-1, which includes both written "self-dialogs" and spoken two-person dialogs, \35Taskmaster-2 consists entirely of spoken two-person dialogs. In addition, while Taskmaster-1 is \36almost exclusively task-based, Taskmaster-2 contains a good number of search- and recommendation-oriented dialogs. \37All dialogs in this release were created using a Wizard of Oz (WOz) methodology in which crowdsourced \38workers played the role of a 'user' and trained call center operators played the role of the 'assistant'. \39In this way, users were led to believe they were interacting with an automated system that “spoke” \40using text-to-speech (TTS) even though it was in fact a human behind the scenes. \41As a result, users could express themselves however they chose in the context of an automated interface.42"""43 44_HOMEPAGE = "https://github.com/google-research-datasets/Taskmaster/tree/master/TM-2-2020"45 46_BASE_URL = "https://raw.githubusercontent.com/google-research-datasets/Taskmaster/master/TM-2-2020/data"47 48 49class Taskmaster2(datasets.GeneratorBasedBuilder):50    """Taskmaster: A dataset for goal oriented conversations."""51 52    VERSION = datasets.Version("1.0.0")53    BUILDER_CONFIGS = [54        datasets.BuilderConfig(55            name="flights", version=datasets.Version("1.0.0"), description="Taskmaster-2 flights domain."56        ),57        datasets.BuilderConfig(58            name="food-ordering", version=datasets.Version("1.0.0"), description="Taskmaster-2 food-ordering domain"59        ),60        datasets.BuilderConfig(61            name="hotels", version=datasets.Version("1.0.0"), description="Taskmaster-2 hotel domain"62        ),63        datasets.BuilderConfig(64            name="movies", version=datasets.Version("1.0.0"), description="Taskmaster-2 movies domain"65        ),66        datasets.BuilderConfig(67            name="music", version=datasets.Version("1.0.0"), description="Taskmaster-2 music domain"68        ),69        datasets.BuilderConfig(70            name="restaurant-search",71            version=datasets.Version("1.0.0"),72            description="Taskmaster-2 restaurant-search domain",73        ),74        datasets.BuilderConfig(75            name="sports", version=datasets.Version("1.0.0"), description="Taskmaster-2 sports domain"76        ),77    ]78 79    def _info(self):80        features = {81            "conversation_id": datasets.Value("string"),82            "instruction_id": datasets.Value("string"),83            "utterances": [84                {85                    "index": datasets.Value("int32"),86                    "speaker": datasets.Value("string"),87                    "text": datasets.Value("string"),88                    "segments": [89                        {90                            "start_index": datasets.Value("int32"),91                            "end_index": datasets.Value("int32"),92                            "text": datasets.Value("string"),93                            "annotations": [{"name": datasets.Value("string")}],94                        }95                    ],96                }97            ],98        }99        return datasets.DatasetInfo(100            description=_DESCRIPTION,101            features=datasets.Features(features),102            supervised_keys=None,103            homepage=_HOMEPAGE,104            citation=_CITATION,105        )106 107    def _split_generators(self, dl_manager):108        url = f"{_BASE_URL}/{self.config.name}.json"109        dialogs_file = dl_manager.download(url)110        return [111            datasets.SplitGenerator(112                name=datasets.Split.TRAIN,113                gen_kwargs={"filepath": dialogs_file},114            ),115        ]116 117    def _generate_examples(self, filepath):118        key = 0119        with open(filepath, encoding="utf-8") as f:120            dialogs = json.load(f)121            for dialog in dialogs:122                utterances = dialog["utterances"]123                for utterance in utterances:124                    if "segments" not in utterance:125                        utterance["segments"] = []126                yield key, dialog127                key += 1128