Team Ai
Apppublic

OSS-forge/CodeQualityEval

sourceHugging Facecc-by-sa-4.0updated 11mo agoView on Hugging Face
12likes
run_semgrep_python.py151 linesDownload Raw Back to 4_Code_Security_Analysis
1import json2import os3import subprocess4import time5import argparse6import shutil7 8CODE_FIELD = os.environ.get("CODE_FIELD", "human_code")9 10FIELD_LABELS = {11    "human_code": "Human",12    "chatgpt_code": "ChatGPT",13    "dsc_code": "DSC",14    "qwen_code": "Qwen",15}16CODE_LABEL = FIELD_LABELS.get(CODE_FIELD, CODE_FIELD)17 18 19def split_jsonl_to_python_files(jsonl_file, output_prefix, lines_per_file=1, files_per_batch=20000):20    start_time = time.time()21 22    outputs = []23 24    with open(jsonl_file, 'r', encoding='utf-8') as f:25        for line in f:26            if line.strip():  # Skip empty lines27                item = json.loads(line)28                try:29                    output = item.get(CODE_FIELD) # Select the key you want to extract30                except Exception:31                    outputs.append(item)32                else:33                    outputs.append(output)34 35    total_lines = len(outputs)36    total_files = total_lines // lines_per_file37    total_batches = (total_files + files_per_batch - 1) // files_per_batch38 39    print(f"Total lines: {total_lines}, Total files: {total_files}, Total batches: {total_batches}")40 41    split_times = []42    semgrep_times = []43    delete_times = []44 45    temp_dir = f"{output_prefix}_tempfiles"46    os.makedirs(temp_dir, exist_ok=True)47 48    for batch in range(total_batches):49        print(f"Processing batch {batch + 1}/{total_batches}")50        batch_start_index = batch * files_per_batch * lines_per_file51        batch_end_index = min((batch + 1) * files_per_batch * lines_per_file, total_lines)52        batch_outputs = outputs[batch_start_index:batch_end_index]53 54        num_files = (batch_end_index - batch_start_index) // lines_per_file55 56        # 1. Write the batch files57        batch_split_start = time.time()58        for i in range(num_files):59            start_index = batch_start_index + i * lines_per_file60            end_index = start_index + lines_per_file61            chunk = batch_outputs[start_index - batch_start_index:end_index - batch_start_index]62 63            output_file = os.path.join(temp_dir, f"{output_prefix}_{start_index+1}.py")64            with open(output_file, 'w', encoding='utf-8') as f_out:65                for line in chunk:66                    f_out.write(line)67        batch_split_end = time.time()68        split_times.append(batch_split_end - batch_split_start)69 70        # 2. Run Semgrep on the batch71        json_filename = f"{output_prefix}_semgrep_results_batch_{batch+1}.json"72        batch_semgrep_time = run_semgrep_analysis(json_filename, temp_dir)73        semgrep_times.append(batch_semgrep_time)74 75        # 3. Clean up only this batch's files76        batch_delete_start = time.time()77        for filename in os.listdir(temp_dir):78            file_path = os.path.join(temp_dir, filename)79            if file_path.endswith('.py') and os.path.isfile(file_path):80                os.remove(file_path)81        batch_delete_end = time.time()82        delete_times.append(batch_delete_end - batch_delete_start)83 84    # Final full clean-up85    shutil.rmtree(temp_dir)86 87    end_time = time.time()88    split_json_time = end_time - start_time89    return split_json_time, split_times, semgrep_times, delete_times90 91def run_semgrep_analysis(json_filename, target_dir):92    start_time = time.time()93 94    print(f"Running Semgrep analysis on {target_dir} and saving results to {json_filename}...")95    semgrep_command = [96        "semgrep", "scan",97        "--verbose",98        "--output", json_filename,99        "--json",100        "--no-git-ignore",101        "--max-memory=30000",102        "--max-target-bytes=1000000",103        "--timeout-threshold", "10",104        "--timeout", "60",105        "--metrics", "off",106        "--include", "*.py",  # <-- only scan Python files107        "--config", "p/trailofbits",108        "--config", "p/default",109        "--config", "p/comment",110        "--config", "p/python",111        "--config", "p/cwe-top-25",112        "--config", "p/owasp-top-ten",113        "--config", "p/r2c-security-audit",114        "--config", "p/insecure-transport",115        "--config", "p/secrets",116        "--config", "p/findsecbugs",117        "--config", "p/gitlab",118        "--config", "p/mobsfscan",119        "--config", "p/command-injection",120        "--config", "p/sql-injection",121        target_dir122    ]123    124    subprocess.run(semgrep_command, check=True)125 126    end_time = time.time()127    run_semgrep_time = end_time - start_time128    return run_semgrep_time129 130if __name__ == "__main__":131    parser = argparse.ArgumentParser(description='Process JSONL file and run Semgrep analysis.')132    parser.add_argument('jsonl_file', type=str, help='The path to the JSONL file.')133 134    args = parser.parse_args()135 136    json_filename = os.path.basename(args.jsonl_file)137    output_prefix = os.path.splitext(json_filename)[0]138 139    start_time = time.time()140 141    split_json_time, split_times, semgrep_times, delete_times = split_jsonl_to_python_files(args.jsonl_file, output_prefix)142 143    end_time = time.time()144    total_time = end_time - start_time145 146    print(f"Total execution time: {total_time:.2f} seconds ({total_time/60:.2f} minutes)")147 148    print("\nDetailed timings per batch:")149    for i, (split_time, semgrep_time, delete_time) in enumerate(zip(split_times, semgrep_times, delete_times), start=1):150        print(f"Batch {i}: Semgrep time: {semgrep_time:.2f} s, Batch cleanup time: {delete_time:.2f} s")151