SPG29/author_profiling
he corpus for the author profiling analysis contains texts in Russian-language which labeled for 5 tasks: 1) gender -- 13530 texts with the labels, who wrote this: text female or male; 2) age -- 13530 texts with the labels, how old the person who wrote the text. This is a number from 12 to 80. In addition, for the classification task we added 5 age groups: 1-19; 20-29; 30-39; 40-49; 50+; 3) age imitation -- 7574 texts, where crowdsource authors is asked to write three texts: a) in their natural manner, b) imitating the style of someone younger, c) imitating the style of someone older; 4) gender imitation -- 5956 texts, where the crowdsource authors is asked to write texts: in their origin gender and pretending to be the opposite gender; 5) style imitation -- 5956 texts, where crowdsource authors is asked to write a text on behalf of another person of your own gender, with a distortion of the authors usual style.
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16# Lint as: python317"""author_profiling dataset"""18 19import json20import os21 22import datasets23 24 25# TODO: Add BibTeX citation26# Find for instance the citation on arxiv or on the dataset repo/website27 28_CITATION = """\29"""30 31_LICENSE = """http://www.apache.org/licenses/LICENSE-2.0"""32 33# TODO: Add description of the dataset here34# You can copy an official description35_DESCRIPTION = """\36he corpus for the author profiling analysis contains texts in Russian-language which labeled for 5 tasks:371) gender -- 13530 texts with the labels, who wrote this: text female or male;382) age -- 13530 texts with the labels, how old the person who wrote the text. This is a number from 12 to 80. In addition, for the classification task we added 5 age groups: 1-19; 20-29; 30-39; 40-49; 50+;393) age imitation -- 7574 texts, where crowdsource authors is asked to write three texts: 40 a) in their natural manner, 41 b) imitating the style of someone younger, 42 c) imitating the style of someone older;434) gender imitation -- 5956 texts, where the crowdsource authors is asked to write texts: in their origin gender and pretending to be the opposite gender;445) style imitation -- 5956 texts, where crowdsource authors is asked to write a text on behalf of another person of your own gender, with a distortion of the authors usual style.45"""46 47# TODO: Add a link to an official homepage for the dataset here48_HOMEPAGE = "https://github.com/sag111/Author-Profiling"49 50# TODO: Add link to the official dataset URLs here51# The HuggingFace dataset library don't host the datasets but only point to the original files52# This can be an arbitrary nested dict/list of URLs (see below in `_split_generators` method)53_URLs = {54 "main": "https://sagteam.ru/author_profiling/main.zip"55}56 57 58# TODO: Name of the dataset usually match the script name with CamelCase instead of snake_case59class AuthorProfiling(datasets.GeneratorBasedBuilder):60 """61 The Corpus for the analysis of author profiling in Russian-language texts.62 """63 64 VERSION = datasets.Version("2.0.1")65 66 # This is an example of a dataset with multiple configurations.67 # If you don't want/need to define several sub-sets in your dataset,68 # just remove the BUILDER_CONFIG_CLASS and the BUILDER_CONFIGS attributes.69 70 # If you need to make complex sub-parts in the datasets with configurable options71 # You can create your own builder configuration class to store attribute, inheriting from datasets.BuilderConfig72 # BUILDER_CONFIG_CLASS = MyBuilderConfig73 74 # You will be able to load one or the other configurations in the following list with75 # data = datasets.load_dataset('my_dataset', 'first_domain')76 # data = datasets.load_dataset('my_dataset', 'second_domain')77 BUILDER_CONFIGS = [78 datasets.BuilderConfig(79 name="main", version=VERSION, description="This a main version of Author Profiling dataset"80 ),81 ]82 83 DEFAULT_CONFIG_NAME = "main" # It's not mandatory to have a default configuration. Just use one if it make sense.84 85 def _info(self):86 # TODO: This method specifies the datasets.DatasetInfo object which contains informations and typings for the dataset87 if self.config.name == "main": # This is the name of the configuration selected in BUILDER_CONFIGS above88 features = datasets.Features(89 {90 "id": datasets.Value("string"),91 "text": datasets.Value("string"),92 "account_id": datasets.Value("string"),93 "author_id": datasets.Value("int64"),94 "age": datasets.Value("int64"),95 "age_group": datasets.Value("string"), #.ClassLabel(names=["0-19", "20-29", "30-39", "40-49", "50+"]),96 "gender": datasets.Value("string"), #.ClassLabel(names=["male", "female"]),97 "no_imitation": datasets.Value("string"), #.ClassLabel(names=["no_any_imitation", "with_any_imitation"]),98 "age_imitation": datasets.Value("string"), #.ClassLabel(names=["no_age_imitation", "younger", "older", "None"]),99 "gender_imitation": datasets.Value("string"), #.ClassLabel(names=["no_gender_imitation", "with_gender_imitation", "None"]),100 "style_imitation": datasets.Value("string"), #.ClassLabel(names=["no_style_imitation", "with_style_imitation", "None"]),101 # These are the features of your dataset like images, labels ...102 }103 )104 else: # This is an example to show how to have different features for "first_domain" and "second_domain"105 pass106 107 return datasets.DatasetInfo(108 # This is the description that will appear on the datasets page.109 description=_DESCRIPTION,110 # This defines the different columns of the dataset and their types111 features=features, # Here we define them above because they are different between the two configurations112 # If there's a common (input, target) tuple from the features,113 # specify them here. They'll be used if as_supervised=True in114 # builder.as_dataset.115 supervised_keys=None,116 # Homepage of the dataset for documentation117 homepage=_HOMEPAGE,118 # License for the dataset if available119 license=_LICENSE,120 # Citation for the dataset121 citation=_CITATION,122 )123 124 def _split_generators(self, dl_manager):125 """Returns SplitGenerators."""126 # TODO: This method is tasked with downloading/extracting the data and defining the splits depending on the configuration127 # If several configurations are possible (listed in BUILDER_CONFIGS), the configuration selected by the user is in self.config.name128 129 # dl_manager is a datasets.download.DownloadManager that can be used to download and extract URLs130 # It can accept any type or nested list/dict and will give back the same structure with the url replaced with path to local files.131 # By default the archives will be extracted and a path to a cached folder where they are extracted is returned instead of the archive132 my_urls = _URLs[self.config.name]133 data_dir = dl_manager.download_and_extract(my_urls)134 return [135 datasets.SplitGenerator(136 name=datasets.Split.TRAIN,137 # These kwargs will be passed to _generate_examples138 gen_kwargs={139 "filepath": os.path.join(data_dir, self.config.name, "train.jsonl"),140 "split": "train",141 },142 ),143 datasets.SplitGenerator(144 name=datasets.Split.VALIDATION,145 # These kwargs will be passed to _generate_examples146 gen_kwargs={147 "filepath": os.path.join(data_dir, self.config.name, "valid.jsonl"),148 "split": "valid",149 },150 ),151 datasets.SplitGenerator(152 name=datasets.Split.TEST,153 # These kwargs will be passed to _generate_examples154 gen_kwargs={155 "filepath": os.path.join(data_dir, self.config.name, "test.jsonl"), 156 "split": "test"157 },158 ),159 ]160 161 def _generate_examples(162 self, filepath, split # method parameters are unpacked from `gen_kwargs` as given in `_split_generators`163 ):164 """Yields examples as (key, example) tuples."""165 # This method handles input defined in _split_generators to yield (key, example) tuples from the dataset.166 # The `key` is here for legacy reason (tfds) and is not important in itself.167 168 with open(filepath, encoding="utf-8") as f:169 for id_, row in enumerate(f):170 data = json.loads(row.rstrip('\n|\r'))171 if self.config.name == "main":172 yield id_, data173 else:174 pass