OSS-forge/CodeQualityEval
12
1import json2import pprint3import argparse4import re5from collections import defaultdict6from collections import defaultdict, Counter 7 8"""9Read filename and max batch number from commandline. Rename all the files to have a single name and number.10"""11 12parser = argparse.ArgumentParser(description='Process Semgrep results.')13parser.add_argument('json_filename', type=str, help='Base filename for Semgrep JSON results')14parser.add_argument('max_batch_num', type=int, help='Maximum batch number to process')15 16args = parser.parse_args()17json_filename = args.json_filename18max_batch_num = args.max_batch_num19 20"""21Read json file in batches and create a single list of total errors, results, and scanned files.22Count number of issues, number of scanned file, number of files that caused errors and compute issues percentage.23 24NB: skipped files contain errors (already accounted for in total_errors) and incompatible rules due to version 25and language (filtered out from the errors)26"""27 28total_errors = []29total_results = []30total_scanned = []31total_skipped = []32 33for i in range(1, max_batch_num + 1):34 json_filename_complete = f"{json_filename}_{i}.json"35 filtered_errors = []36 with open(json_filename_complete, 'r', encoding='utf-8') as results_f:37 samples = json.load(results_f)38 filtered_errors.extend(39 [error for error in samples['errors'] if not error['path'].startswith('https:/semgrep.dev/...')]40 ) # Filtering out incompatible rules41 total_errors.extend(filtered_errors)42 total_results.extend(samples['results'])43 total_scanned.extend(samples['paths']['scanned'])44 total_skipped.extend(samples['paths']['skipped'])45 46"""47Calculate file number from the filename to obtain the dataset line number and insert it into the path field.48This is done to filter out duplicates.49"""50pattern = r'TempClass(\d+)\.java'51 52def calculate_line_number(filename):53 match = re.match(pattern, filename)54 return int(match.group(1)) if match else None55 56 57for error in total_errors:58 error['path'] = calculate_line_number(error['path'])59 60for result in total_results:61 result['path'] = calculate_line_number(result['path'])62 63for i in range(len(total_scanned)):64 total_scanned[i] = calculate_line_number(total_scanned[i])65 66 67"""68Remove duplicates from the errors and results lists.69_____________________70dedup_err is the list of errors w/o duplicates71dedup_res is the list of defective functions (i.e., w/o duplicated issues)72total_results is the list of issues w/o errors73dedup_res_no_errors is the list of defective functions w/o errors74"""75 76dedup_err = {err['path'] for err in total_errors}77dedup_res = {res['path'] for res in total_results}78 79dedup_res_no_errors = [res for res in dedup_res if res not in dedup_err]80total_results = [res for res in total_results if res['path'] not in dedup_err]81 82"""83Normalize CWE names dynamically to ensure uniqueness.84"""85 86def extract_cwe_number(cwe_name):87 """Extract CWE-XXX format from any given CWE description."""88 match = re.match(r"(CWE-\d+)", cwe_name, re.IGNORECASE)89 return match.group(1) if match else cwe_name90 91 92"""93Divide issues based on category type. 94Since not all issues are correctly categories (i.e., missing "category" field), 95we select them based on whether they have a "CWE" field.96"""97 98security_issues = []99seen_issues = set()100severity_types = set()101normalized_cwe_dict = defaultdict(str)102 103# Process security issues and normalize CWEs104for result in total_results:105 metadata = result.get('extra', {}).get('metadata', {})106 cwes = metadata.get('cwe')107 severity = result.get('extra', {}).get('severity')108 109 if cwes:110 if isinstance(cwes, list):111 updated_cwes = []112 for cwe in cwes:113 base_cwe = extract_cwe_number(cwe)114 if base_cwe in normalized_cwe_dict:115 standardized_cwe = max(normalized_cwe_dict[base_cwe], cwe, key=len)116 else:117 standardized_cwe = cwe # Keep first occurrence as reference118 normalized_cwe_dict[base_cwe] = standardized_cwe119 updated_cwes.append(standardized_cwe)120 result['extra']['metadata']['cwe'] = [cwe.upper() for cwe in updated_cwes]121 else:122 cwes = f"{cwes.upper()}"123 base_cwe = extract_cwe_number(cwes)124 if base_cwe in normalized_cwe_dict:125 standardized_cwe = max(normalized_cwe_dict[base_cwe], cwes, key=len)126 else:127 standardized_cwe = cwes # Keep first occurrence as reference128 normalized_cwe_dict[base_cwe] = standardized_cwe129 result['extra']['metadata']['cwe'] = standardized_cwe.upper()130 131 # Use a unique identifier for each issue (path, CWE, severity, and message)132 issue_id = (133 result['path'], 134 tuple(sorted(result['extra']['metadata']['cwe'])), # Ensure consistent ordering of CWEs135 result['extra'].get('severity', ''), 136 result['extra'].get('lines', '').strip(), # Remove accidental whitespace137 )138 139 if issue_id not in seen_issues:140 seen_issues.add(issue_id) # Add to set to track unique issues141 security_issues.append(result)142 143 if severity:144 severity_types.add(severity)145 146# Deduplicate CWEs by keeping only the longest description for each CWE number147deduplicated_cwes = {}148 149for base_cwe, cwe_description in normalized_cwe_dict.items():150 base_cwe = base_cwe.upper() # Ensure "CWE" is always uppercase151 cwe_description = cwe_description.strip() # Remove any accidental spaces152 153 # Keep the longest description per CWE number154 if base_cwe not in deduplicated_cwes or len(cwe_description) > len(deduplicated_cwes[base_cwe]):155 deduplicated_cwes[base_cwe] = cwe_description156 157unified_cwes = set(deduplicated_cwes.values())158 159for result in security_issues:160 metadata = result.get('extra', {}).get('metadata', {})161 cwes = metadata.get('cwe')162 163 if cwes:164 if isinstance(cwes, list):165 result['extra']['metadata']['cwe'] = [deduplicated_cwes[extract_cwe_number(cwe).upper()] for cwe in cwes]166 else:167 result['extra']['metadata']['cwe'] = deduplicated_cwes[extract_cwe_number(cwes).upper()]168 169"""170NEW: Compute and print the Top‑10 most frequent CWEs across the dataset171"""172 173cwe_counter = Counter()174for issue in security_issues:175 cwes = issue['extra']['metadata']['cwe']176 if isinstance(cwes, list):177 cwe_counter.update(cwes)178 else:179 cwe_counter.update([cwes])180 181 182"""183Divide security-related issues by CWE severity category.184"""185 186cwes_by_severity = {severity: {} for severity in severity_types}187 188for issue in security_issues:189 metadata = issue.get('extra', {}).get('metadata', {})190 cwes = metadata.get('cwe')191 severity = issue.get('extra', {}).get('severity')192 193 if severity and cwes:194 if isinstance(cwes, list):195 for cwe in cwes:196 if cwe not in cwes_by_severity[severity]:197 cwes_by_severity[severity][cwe] = []198 cwes_by_severity[severity][cwe].append(issue)199 else:200 if cwes not in cwes_by_severity[severity]:201 cwes_by_severity[severity][cwes] = []202 cwes_by_severity[severity][cwes].append(issue)203 204cwes_counts_by_severity = {205 severity: {cwe: len(issues) for cwe, issues in cwes_dict.items()}206 for severity, cwes_dict in cwes_by_severity.items()207}208 209"""210Compute percentages of defects, errors and clean functions.211 212NB: security_issues is already error-free because "total_results" is error free 213-> we only need to remove path duplicates to obtain the number of defective functions (only security)214"""215 216# Computing defective functions (i.e., removing duplicate security issues). 217# We only need the number and path to later remove them from the dataset218defective_func_security_set = {issue['path'] for issue in security_issues}219 220defective_func_rate = (len(defective_func_security_set) / len(total_scanned)) * 100221errors_rate = (len(dedup_err) / len(total_scanned)) * 100222clean_rate = ((len(total_scanned) - len(defective_func_security_set) - len(dedup_err)) / len(total_scanned)) * 100223 224print(f"Total skipped functions: {len(total_skipped)} (errors + incompatible rules)")225print(f"Total scanned functions: {len(total_scanned)} (100%)")226print(f"Total clean functions: {len(total_scanned)-len(defective_func_security_set)-len(dedup_err)} ({clean_rate:.2f}%)")227print(f"Total defective functions (excluding errors): {len(defective_func_security_set)} ({defective_func_rate:.2f}%)")228print(f"Total errors: {len(total_errors)}. Errors w/o duplicates: {len(dedup_err)} ({errors_rate:.2f}%)")229print(f"Total issues (considering multiple issues per function and excluding errors): {len(security_issues)}")230 231print(f"\nFinal Unified CWE Set (without duplicates): {len(unified_cwes)}")232# pprint.pprint(unified_cwes)233 234print("\nTop 10 CWEs by occurrence (across all severities):")235for rank, (cwe, count) in enumerate(cwe_counter.most_common(10), start=1):236 print(f"{rank:2}. {cwe}: {count}")237 238print(f"\nSeverity types: {severity_types}")239print(f"CWEs divided by severity:")240pprint.pprint(cwes_counts_by_severity)