Team Ai
Datasetpublic

rombodawg/data_processing_code

These are the scripts I used to clean the rombodawg/code_bagel and rombodawg/code_bagel_hermes-2.5 datasets. In order for these scripts to work your datasets need to be in the format bellow, with the condition that each line is its own .json object. {"instruction": "", "input": "", "output": ""} {"instruction": "", "input": "", "output": ""} {"instruction": "", "input": "", "output": ""} {"instruction": "", "input": "", "output": ""} {"instruction": "", "input": "", "output": ""}

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
4likes19downloads
dedupe.py61 linesDownload Raw Back to root
1import json
2import os
3from typing import Iterator
4from concurrent.futures import ThreadPoolExecutor, as_completed
5
6def remove_duplicate_outputs(input_file_name: str, output_file_name: str, num_threads: int = 10) -> None:
7    """
8    Removes duplicate entries in the "output" field from the input file and writes the filtered lines to the output file.
9
10    :param input_file_name: Name of the input JSON file.
11    :param output_file_name: Name of the output JSON file.
12    :param num_threads: Number of threads to use for parallel processing (defaults to 10).
13    """
14    script_dir = os.path.dirname(os.path.abspath(__file__))
15    input_file_path = os.path.join(script_dir, input_file_name)
16    output_file_path = os.path.join(script_dir, output_file_name)
17
18    with open(input_file_path, 'r', encoding='utf-8') as input_fp:
19        lines = input_fp.readlines()
20
21    with ThreadPoolExecutor(max_workers=num_threads) as executor:
22        futures = [executor.submit(load_and_filter_lines, lines) for _ in range(num_threads)]
23        filtered_lines = []
24        for future in as_completed(futures):
25            filtered_lines.extend(future.result())
26
27    seen_outputs = set()
28    with open(output_file_path, 'w', encoding='utf-8') as output_fp:
29        for line in filtered_lines:
30            try:
31                data = json.loads(line)
32                output_value = data.get('output', '')
33                if output_value not in seen_outputs:
34                    seen_outputs.add(output_value)
35                    output_fp.write(line)
36            except json.JSONDecodeError:
37                continue
38
39def load_and_filter_lines(lines: list[str]) -> list[str]:
40    """
41    Loads and filters lines from the input dataset.
42
43    :param lines: List of lines from the input dataset.
44    :return: List of filtered lines.
45    """
46    filtered_lines = []
47    for line in lines:
48        try:
49            data = json.loads(line)
50            output_value = data.get('output', '')
51            if output_value:
52                filtered_lines.append(line)
53        except json.JSONDecodeError:
54            continue
55    return filtered_lines
56
57# Here is where you put the name of the input file, and name of the file that will be created. Note the input file needs to be in the same directory as the script unless you edit the code to have a file path instead.
58input_file_name = 'Input_File.json'
59output_file_name = 'Output_File.json'
60
61remove_duplicate_outputs(input_file_name, output_file_name, num_threads=10)