OSS-forge/CodeQualityEval
12
1import os2import shutil3import subprocess4import textwrap5 6import gradio as gr7 8ROOT_DIR = os.path.dirname(os.path.abspath(__file__))9 10CODE_FIELDS = ["human_code", "chatgpt_code", "dsc_code", "qwen_code"]11 12 13def run_command(cmd: str, cwd: str | None = None, extra_env: dict | None = None) -> str:14 """15 Run a shell command, capture stdout+stderr and return them as text.16 """17 if cwd is None:18 cwd = ROOT_DIR19 20 env = os.environ.copy()21 if extra_env:22 env.update(extra_env)23 24 try:25 result = subprocess.run(26 cmd,27 shell=True,28 cwd=cwd,29 env=env,30 stdout=subprocess.PIPE,31 stderr=subprocess.STDOUT,32 text=True,33 )34 return f"$ {cmd}\n\n{result.stdout}"35 except Exception as e:36 return f"$ {cmd}\n\nERROR: {e}"37 38 39# ----------------------------40# PYTHON ANALYSES41# ----------------------------42 43def run_python_defects(code_field: str) -> str:44 """45 Run Python defects analysis on the sample dataset for the selected field:46 - pylint_ODC.py47 - process_pylint_results.py48 """49 logs = []50 logs.append(51 f"### Running Python defects analysis on sample dataset ({code_field})\n"52 )53 54 env = {"CODE_FIELD": code_field}55 56 logs.append(57 run_command(58 "python 3_Code_Defects_Analysis/pylint_ODC.py",59 extra_env=env,60 )61 )62 logs.append(63 run_command(64 "python 3_Code_Defects_Analysis/process_pylint_results.py",65 extra_env=env,66 )67 )68 69 return "\n\n".join(logs)70 71 72def run_python_security(code_field: str) -> str:73 """74 Run Python security analysis on the sample dataset for the selected field:75 - run_semgrep_python.py76 - process_semgrep_results_python.py77 """78 logs = []79 logs.append(80 f"### Running Python security (Semgrep) analysis on sample dataset ({code_field})\n"81 )82 83 env = {"CODE_FIELD": code_field}84 85 logs.append(86 run_command(87 "python 4_Code_Security_Analysis/run_semgrep_python.py "88 "1_dataset_sample_100/python_dataset.jsonl",89 extra_env=env,90 )91 )92 logs.append(93 run_command(94 "python 4_Code_Security_Analysis/process_semgrep_results_python.py "95 "python_dataset_semgrep_results_batch 1",96 extra_env=env,97 )98 )99 100 return "\n\n".join(logs)101 102 103def run_python_complexity() -> str:104 """105 Run Python complexity analysis on the sample dataset.106 (Complexity runs on all features together – no CODE_FIELD.)107 """108 logs = []109 logs.append("### Running Python complexity analysis on sample dataset (all code fields)\n")110 logs.append(111 run_command(112 "python 5_Code_Complexity_Analysis/complexity_stats_python.py",113 )114 )115 return "\n\n".join(logs)116 117 118# ----------------------------119# JAVA ANALYSES120# ----------------------------121 122def run_java_defects(code_field: str) -> str:123 """124 Run Java defects analysis on the sample dataset for the selected field:125 - wrap_java_functions.py126 - run_PMD_analysis.sh127 - process_PMD_results.py128 """129 logs = []130 logs.append(131 f"### Running Java defects analysis on sample dataset ({code_field})\n"132 )133 134 env = {"CODE_FIELD": code_field}135 136 # fresh temp directory for wrapped .java files137 temp_dir = os.path.join(ROOT_DIR, "java_temp_wrapped")138 if os.path.exists(temp_dir):139 shutil.rmtree(temp_dir)140 logs.append(run_command(f"mkdir -p {temp_dir}"))141 142 # Wrap Java functions for the selected code field143 # (script reads CODE_FIELD from env; CLI arg is kept for compatibility)144 logs.append(145 run_command(146 "python 3_Code_Defects_Analysis/wrap_java_functions.py "147 "1_dataset_sample_100/java_dataset.jsonl",148 extra_env=env,149 )150 )151 152 # Run PMD analysis script on the wrapped folder153 logs.append(154 run_command(155 "bash 3_Code_Defects_Analysis/run_PMD_analysis.sh java_temp_wrapped",156 )157 )158 159 # Organize PMD results160 pmd_human_dir = os.path.join(ROOT_DIR, "PMD_Human")161 logs.append(run_command("mkdir -p PMD_Human"))162 logs.append(run_command("mkdir -p reports errors", cwd=pmd_human_dir))163 logs.append(run_command("mv ../report_unique_* reports || true", cwd=pmd_human_dir))164 logs.append(run_command("mv ../errors_unique_* errors || true", cwd=pmd_human_dir))165 166 # Process PMD results (script can use CODE_FIELD to choose output filenames)167 logs.append(168 run_command(169 "python ../3_Code_Defects_Analysis/process_PMD_results.py",170 cwd=pmd_human_dir,171 extra_env=env,172 )173 )174 175 return "\n\n".join(logs)176 177 178def run_java_security(code_field: str) -> str:179 """180 Run Java security analysis on the sample dataset for the selected field:181 - run_semgrep_java.py182 - process_semgrep_results_java.py183 """184 logs = []185 logs.append(186 f"### Running Java security (Semgrep) analysis on sample dataset ({code_field})\n"187 )188 189 env = {"CODE_FIELD": code_field}190 191 logs.append(192 run_command(193 "python 4_Code_Security_Analysis/run_semgrep_java.py "194 "1_dataset_sample_100/java_dataset.jsonl 100",195 extra_env=env,196 )197 )198 logs.append(199 run_command(200 "python 4_Code_Security_Analysis/process_semgrep_results_java.py "201 "semgrep_batches/1_dataset_sample_100/java_dataset.jsonl_semgrep_results_batch 1",202 extra_env=env,203 )204 )205 206 return "\n\n".join(logs)207 208 209def run_java_complexity() -> str:210 """211 Run Java complexity analysis on the sample dataset.212 (Complexity runs on all features together – no CODE_FIELD.)213 """214 logs = []215 logs.append("### Running Java complexity analysis on sample dataset (all code fields)\n")216 logs.append(217 run_command(218 "python 5_Code_Complexity_Analysis/complexity_stats_java.py",219 )220 )221 return "\n\n".join(logs)222 223 224# ----------------------------225# GRADIO UI226# ----------------------------227 228intro_md = textwrap.dedent(229 """230 # Code Quality Evaluation: Human-written vs. AI-generated231 232 This Space can run the following analyses on Python and Java code:233 234 - **Defects** (Pylint for Python, PMD for Java + ODC mapping)235 - **Security vulnerabilities** (Semgrep for Python & Java)236 - **Complexity** (Lizard + Tiktoken for Python & Java)237 238 All runs here use the **sample dataset (100 instances)** for reproducibility and speed. Refer to the paper for the complete dataset.239 240 You can choose which code field to analyze for **defects** and **security**:241 - `human_code`242 - `chatgpt_code`243 - `dsc_code`244 - `qwen_code`245 246 Complexity analyses run over all code fields together.247 """248)249 250 251with gr.Blocks() as demo:252 gr.Markdown(intro_md)253 254 # Global selector for which dataset field to analyze255 code_field_dropdown = gr.Dropdown(256 label="Dataset code field (for defects & security)",257 choices=CODE_FIELDS,258 value="human_code",259 )260 261 with gr.Tab("Python"):262 gr.Markdown("## Python Analyses")263 264 with gr.Row():265 with gr.Column():266 btn_py_defects = gr.Button("Run Python Defects Analysis")267 btn_py_security = gr.Button("Run Python Security Analysis")268 btn_py_complexity = gr.Button("Run Python Complexity Analysis")269 270 with gr.Column():271 out_py_defects = gr.Textbox(272 label="Python Defects Output",273 lines=20,274 )275 out_py_security = gr.Textbox(276 label="Python Security Output",277 lines=20,278 )279 out_py_complexity = gr.Textbox(280 label="Python Complexity Output",281 lines=20,282 )283 284 # Defects & security depend on CODE_FIELD285 btn_py_defects.click(286 run_python_defects, inputs=code_field_dropdown, outputs=out_py_defects287 )288 btn_py_security.click(289 run_python_security, inputs=code_field_dropdown, outputs=out_py_security290 )291 # Complexity runs on all fields together – no CODE_FIELD input292 btn_py_complexity.click(293 run_python_complexity, outputs=out_py_complexity294 )295 296 with gr.Tab("Java"):297 gr.Markdown("## Java Analyses")298 299 with gr.Row():300 with gr.Column():301 btn_java_defects = gr.Button("Run Java Defects Analysis")302 btn_java_security = gr.Button("Run Java Security Analysis")303 btn_java_complexity = gr.Button("Run Java Complexity Analysis")304 305 with gr.Column():306 out_java_defects = gr.Textbox(307 label="Java Defects Output",308 lines=20,309 )310 out_java_security = gr.Textbox(311 label="Java Security Output",312 lines=20,313 )314 out_java_complexity = gr.Textbox(315 label="Java Complexity Output",316 lines=20,317 )318 319 # Defects & security depend on CODE_FIELD320 btn_java_defects.click(321 run_java_defects, inputs=code_field_dropdown, outputs=out_java_defects322 )323 btn_java_security.click(324 run_java_security, inputs=code_field_dropdown, outputs=out_java_security325 )326 # Complexity runs on all fields together – no CODE_FIELD input327 btn_java_complexity.click(328 run_java_complexity, outputs=out_java_complexity329 )330 331 with gr.Tab("About"):332 gr.Markdown(333 """334 ### Notes335 336 - This UI runs the same scripts as described in the artifact:337 - `3_Code_Defects_Analysis/pylint_ODC.py` + `process_pylint_results.py`338 - `3_Code_Defects_Analysis/wrap_java_functions.py` + `run_PMD_analysis.sh` + `process_PMD_results.py`339 - `4_Code_Security_Analysis/run_semgrep_python.py` / `run_semgrep_java.py` + processing scripts340 - `5_Code_Complexity_Analysis/complexity_stats_python.py` / `complexity_stats_java.py`341 - The selected **Dataset code field** (e.g., `human_code`, `chatgpt_code`, `dsc_code`, `qwen_code`)342 is passed to the defects and security scripts via the `CODE_FIELD` environment variable.343 - Complexity analyses remain unchanged from the original artifact and run across all code fields.344 """345 )346 347 348if __name__ == "__main__":349 demo.launch()350 