OSS-forge/CodeQualityEval
12
1import json2import os3import subprocess4import time5import argparse6import shutil7 8CODE_FIELD = os.environ.get("CODE_FIELD", "human_code")9 10FIELD_LABELS = {11 "human_code": "Human",12 "chatgpt_code": "ChatGPT",13 "dsc_code": "DSC",14 "qwen_code": "Qwen",15}16CODE_LABEL = FIELD_LABELS.get(CODE_FIELD, CODE_FIELD)17 18 19def split_jsonl_to_python_files(jsonl_file, output_prefix, lines_per_file=1, files_per_batch=20000):20 start_time = time.time()21 22 outputs = []23 24 with open(jsonl_file, 'r', encoding='utf-8') as f:25 for line in f:26 if line.strip(): # Skip empty lines27 item = json.loads(line)28 try:29 output = item.get(CODE_FIELD) # Select the key you want to extract30 except Exception:31 outputs.append(item)32 else:33 outputs.append(output)34 35 total_lines = len(outputs)36 total_files = total_lines // lines_per_file37 total_batches = (total_files + files_per_batch - 1) // files_per_batch38 39 print(f"Total lines: {total_lines}, Total files: {total_files}, Total batches: {total_batches}")40 41 split_times = []42 semgrep_times = []43 delete_times = []44 45 temp_dir = f"{output_prefix}_tempfiles"46 os.makedirs(temp_dir, exist_ok=True)47 48 for batch in range(total_batches):49 print(f"Processing batch {batch + 1}/{total_batches}")50 batch_start_index = batch * files_per_batch * lines_per_file51 batch_end_index = min((batch + 1) * files_per_batch * lines_per_file, total_lines)52 batch_outputs = outputs[batch_start_index:batch_end_index]53 54 num_files = (batch_end_index - batch_start_index) // lines_per_file55 56 # 1. Write the batch files57 batch_split_start = time.time()58 for i in range(num_files):59 start_index = batch_start_index + i * lines_per_file60 end_index = start_index + lines_per_file61 chunk = batch_outputs[start_index - batch_start_index:end_index - batch_start_index]62 63 output_file = os.path.join(temp_dir, f"{output_prefix}_{start_index+1}.py")64 with open(output_file, 'w', encoding='utf-8') as f_out:65 for line in chunk:66 f_out.write(line)67 batch_split_end = time.time()68 split_times.append(batch_split_end - batch_split_start)69 70 # 2. Run Semgrep on the batch71 json_filename = f"{output_prefix}_semgrep_results_batch_{batch+1}.json"72 batch_semgrep_time = run_semgrep_analysis(json_filename, temp_dir)73 semgrep_times.append(batch_semgrep_time)74 75 # 3. Clean up only this batch's files76 batch_delete_start = time.time()77 for filename in os.listdir(temp_dir):78 file_path = os.path.join(temp_dir, filename)79 if file_path.endswith('.py') and os.path.isfile(file_path):80 os.remove(file_path)81 batch_delete_end = time.time()82 delete_times.append(batch_delete_end - batch_delete_start)83 84 # Final full clean-up85 shutil.rmtree(temp_dir)86 87 end_time = time.time()88 split_json_time = end_time - start_time89 return split_json_time, split_times, semgrep_times, delete_times90 91def run_semgrep_analysis(json_filename, target_dir):92 start_time = time.time()93 94 print(f"Running Semgrep analysis on {target_dir} and saving results to {json_filename}...")95 semgrep_command = [96 "semgrep", "scan",97 "--verbose",98 "--output", json_filename,99 "--json",100 "--no-git-ignore",101 "--max-memory=30000",102 "--max-target-bytes=1000000",103 "--timeout-threshold", "10",104 "--timeout", "60",105 "--metrics", "off",106 "--include", "*.py", # <-- only scan Python files107 "--config", "p/trailofbits",108 "--config", "p/default",109 "--config", "p/comment",110 "--config", "p/python",111 "--config", "p/cwe-top-25",112 "--config", "p/owasp-top-ten",113 "--config", "p/r2c-security-audit",114 "--config", "p/insecure-transport",115 "--config", "p/secrets",116 "--config", "p/findsecbugs",117 "--config", "p/gitlab",118 "--config", "p/mobsfscan",119 "--config", "p/command-injection",120 "--config", "p/sql-injection",121 target_dir122 ]123 124 subprocess.run(semgrep_command, check=True)125 126 end_time = time.time()127 run_semgrep_time = end_time - start_time128 return run_semgrep_time129 130if __name__ == "__main__":131 parser = argparse.ArgumentParser(description='Process JSONL file and run Semgrep analysis.')132 parser.add_argument('jsonl_file', type=str, help='The path to the JSONL file.')133 134 args = parser.parse_args()135 136 json_filename = os.path.basename(args.jsonl_file)137 output_prefix = os.path.splitext(json_filename)[0]138 139 start_time = time.time()140 141 split_json_time, split_times, semgrep_times, delete_times = split_jsonl_to_python_files(args.jsonl_file, output_prefix)142 143 end_time = time.time()144 total_time = end_time - start_time145 146 print(f"Total execution time: {total_time:.2f} seconds ({total_time/60:.2f} minutes)")147 148 print("\nDetailed timings per batch:")149 for i, (split_time, semgrep_time, delete_time) in enumerate(zip(split_times, semgrep_times, delete_times), start=1):150 print(f"Batch {i}: Semgrep time: {semgrep_time:.2f} s, Batch cleanup time: {delete_time:.2f} s")151 