Team Ai
Datasetpublic

SetFit/ethos_binary

This is the binary split of ethos, split into train and test. It contains comments annotated for hate speech or not.

sourceHugging Faceupdated 5y agoView on Hugging Face
1likes336downloads
prepare_data.py22 linesDownload Raw Back to root
1from datasets import load_dataset
2import json
3import random
4
5dataset = load_dataset("ethos", "binary")
6id2label = dataset['train'].features['label'].names
7
8rows = [{'text': row['text'], 'label': row['label'], 'label_text': id2label[row['label']].replace("_", " ")} for row in dataset['train']]
9
10random.seed(42)
11random.shuffle(rows)
12
13num_test = 400
14
15datasplits = {'train': rows[num_test:],  'test': rows[0:num_test]}
16
17
18for split in ['train', 'test']:
19    with open(f'{split}.jsonl', 'w') as fOut:
20        for row in datasplits[split]:
21            fOut.write(json.dumps(row)+"\n")
22