SetFit/ethos_binary
This is the binary split of ethos, split into train and test. It contains comments annotated for hate speech or not.
1336
1from datasets import load_dataset
2import json
3import random
4
5dataset = load_dataset("ethos", "binary")
6id2label = dataset['train'].features['label'].names
7
8rows = [{'text': row['text'], 'label': row['label'], 'label_text': id2label[row['label']].replace("_", " ")} for row in dataset['train']]
9
10random.seed(42)
11random.shuffle(rows)
12
13num_test = 400
14
15datasplits = {'train': rows[num_test:], 'test': rows[0:num_test]}
16
17
18for split in ['train', 'test']:
19 with open(f'{split}.jsonl', 'w') as fOut:
20 for row in datasplits[split]:
21 fOut.write(json.dumps(row)+"\n")
22 