OSS-forge/CodeQualityEval
12
1import json2import pprint3import argparse4import re5from collections import defaultdict6from collections import defaultdict, Counter 7 8"""9Read filename and max batch number from commandline. Rename all the files to have a single name and number.10"""11 12parser = argparse.ArgumentParser(description='Process Semgrep results.')13parser.add_argument('json_filename', type=str, help='Base filename for Semgrep JSON results')14parser.add_argument('max_batch_num', type=int, help='Maximum batch number to process')15 16args = parser.parse_args()17json_filename = args.json_filename18max_batch_num = args.max_batch_num19 20"""21Read json file in batches and create a single list of total errors, results, and scanned files.22Count number of issues, number of scanned file, number of files that caused errors and compute issues percentage.23 24NB: skipped files contain errors (already accounted for in total_errors) and incompatible rules due to version 25and language (filtered out from the errors)26"""27 28total_errors = []29total_results = []30total_scanned = []31total_skipped = []32 33for i in range(1, max_batch_num + 1):34 json_filename_complete = f"{json_filename}_{i}.json"35 filtered_errors = []36 with open(json_filename_complete, 'r', encoding='utf-8') as results_f:37 samples = json.load(results_f)38 filtered_errors.extend(39 [error for error in samples['errors'] if not error['path'].startswith('https:/semgrep.dev/...')]40 ) # Filtering out incompatible rules41 total_errors.extend(filtered_errors)42 total_results.extend(samples['results'])43 total_scanned.extend(samples['paths']['scanned'])44 total_skipped.extend(samples['paths']['skipped'])45 46"""47Calculate file number from the filename to obtain the dataset line number and insert it into the path field.48This is done to filter out duplicates.49"""50pattern = r'.*_(\d+)\.py'51 52def calculate_line_number(filename):53 match = re.match(pattern, filename)54 # print(f"Filename: {filename}, Match: {int(match.group(1)) if match else None}")55 return int(match.group(1)) if match else None56 57 58for error in total_errors:59 error['path'] = calculate_line_number(error['path'])60 61for result in total_results:62 result['path'] = calculate_line_number(result['path'])63 64for i in range(len(total_scanned)):65 total_scanned[i] = calculate_line_number(total_scanned[i])66 67 68"""69Remove duplicates from the errors and results lists.70_______________________71dedup_err is the list of errors w/o duplicates72dedup_res is the list of defective functions (i.e., w/o duplicated issues)73total_results is the list of issues w/o errors74dedup_res_no_errors is the list of defective functions w/o errors75"""76 77dedup_err = {err['path'] for err in total_errors}78dedup_res = {res['path'] for res in total_results}79 80dedup_res_no_errors = [res for res in dedup_res if res not in dedup_err]81total_results = [res for res in total_results if res['path'] not in dedup_err]82 83"""84Normalize CWE names dynamically to ensure uniqueness.85"""86 87def extract_cwe_number(cwe_name):88 """Extract CWE-XXX format from any given CWE description."""89 match = re.match(r"(CWE-\d+)", cwe_name, re.IGNORECASE)90 return match.group(1) if match else cwe_name91 92 93"""94Divide issues based on category type. 95Since not all issues are correctly categories (i.e., missing "category" field), 96we select them based on whether they have a "CWE" field.97"""98 99security_issues = []100seen_issues = set()101severity_types = set()102normalized_cwe_dict = defaultdict(str)103 104# Process security issues and normalize CWEs105for result in total_results:106 metadata = result.get('extra', {}).get('metadata', {})107 cwes = metadata.get('cwe')108 severity = result.get('extra', {}).get('severity')109 110 if cwes:111 if isinstance(cwes, list):112 updated_cwes = []113 for cwe in cwes:114 base_cwe = extract_cwe_number(cwe)115 if base_cwe in normalized_cwe_dict:116 standardized_cwe = max(normalized_cwe_dict[base_cwe], cwe, key=len)117 else:118 standardized_cwe = cwe # Keep first occurrence as reference119 normalized_cwe_dict[base_cwe] = standardized_cwe120 updated_cwes.append(standardized_cwe)121 result['extra']['metadata']['cwe'] = [cwe.upper() for cwe in updated_cwes]122 else:123 cwes = f"{cwes.upper()}"124 base_cwe = extract_cwe_number(cwes)125 if base_cwe in normalized_cwe_dict:126 standardized_cwe = max(normalized_cwe_dict[base_cwe], cwes, key=len)127 else:128 standardized_cwe = cwes # Keep first occurrence as reference129 normalized_cwe_dict[base_cwe] = standardized_cwe130 result['extra']['metadata']['cwe'] = standardized_cwe.upper()131 132 # Use a unique identifier for each issue (path, CWE, severity, and message)133 issue_id = (134 result['path'], 135 tuple(sorted(result['extra']['metadata']['cwe'])), # Ensure consistent ordering of CWEs136 result['extra'].get('severity', ''), 137 result['extra'].get('lines', '').strip(), # Remove accidental whitespace138 )139 140 if issue_id not in seen_issues:141 seen_issues.add(issue_id) # Add to set to track unique issues142 security_issues.append(result)143 144 if severity:145 severity_types.add(severity)146 147# Deduplicate CWEs by keeping only the longest description for each CWE number148deduplicated_cwes = {}149 150for base_cwe, cwe_description in normalized_cwe_dict.items():151 base_cwe = base_cwe.upper() # Ensure "CWE" is always uppercase152 cwe_description = cwe_description.strip() # Remove any accidental spaces153 154 # Keep the longest description per CWE number155 if base_cwe not in deduplicated_cwes or len(cwe_description) > len(deduplicated_cwes[base_cwe]):156 deduplicated_cwes[base_cwe] = cwe_description157 158unified_cwes = set(deduplicated_cwes.values())159 160for result in security_issues:161 metadata = result.get('extra', {}).get('metadata', {})162 cwes = metadata.get('cwe')163 164 if cwes:165 if isinstance(cwes, list):166 result['extra']['metadata']['cwe'] = [deduplicated_cwes[extract_cwe_number(cwe).upper()] for cwe in cwes]167 else:168 result['extra']['metadata']['cwe'] = deduplicated_cwes[extract_cwe_number(cwes).upper()]169 170"""171NEW: Compute and print the Top‑10 most frequent CWEs across the dataset172"""173 174cwe_counter = Counter()175for issue in security_issues:176 cwes = issue['extra']['metadata']['cwe']177 if isinstance(cwes, list):178 cwe_counter.update(cwes)179 else:180 cwe_counter.update([cwes])181 182 183"""184Divide security-related issues by CWE severity category.185"""186 187cwes_by_severity = {severity: {} for severity in severity_types}188 189for issue in security_issues:190 metadata = issue.get('extra', {}).get('metadata', {})191 cwes = metadata.get('cwe')192 severity = issue.get('extra', {}).get('severity')193 194 if severity and cwes:195 if isinstance(cwes, list):196 for cwe in cwes:197 if cwe not in cwes_by_severity[severity]:198 cwes_by_severity[severity][cwe] = []199 cwes_by_severity[severity][cwe].append(issue)200 else:201 if cwes not in cwes_by_severity[severity]:202 cwes_by_severity[severity][cwes] = []203 cwes_by_severity[severity][cwes].append(issue)204 205cwes_counts_by_severity = {206 severity: {cwe: len(issues) for cwe, issues in cwes_dict.items()}207 for severity, cwes_dict in cwes_by_severity.items()208}209 210"""211Compute percentages of defects, errors and clean functions.212 213NB: security_issues is already error-free because "total_results" is error free 214-> we only need to remove path duplicates to obtain the number of defective functions (only security)215"""216 217# Computing defective functions (i.e., removing duplicate security issues). 218# We only need the number and path to later remove them from the dataset219defective_func_security_set = {issue['path'] for issue in security_issues}220 221defective_func_rate = (len(defective_func_security_set) / len(total_scanned)) * 100222errors_rate = (len(dedup_err) / len(total_scanned)) * 100223clean_rate = ((len(total_scanned) - len(defective_func_security_set) - len(dedup_err)) / len(total_scanned)) * 100224 225print(f"Total skipped functions: {len(total_skipped)} (errors + incompatible rules)")226print(f"Total scanned functions: {len(total_scanned)} (100%)")227print(f"Total clean functions: {len(total_scanned)-len(defective_func_security_set)-len(dedup_err)} ({clean_rate:.2f}%)")228print(f"Total defective functions (excluding errors): {len(defective_func_security_set)} ({defective_func_rate:.2f}%)")229print(f"Total errors: {len(total_errors)}. Errors w/o duplicates: {len(dedup_err)} ({errors_rate:.2f}%)")230print(f"Total issues (considering multiple issues per function and excluding errors): {len(security_issues)}")231 232print(f"\nFinal Unified CWE Set (without duplicates): {len(unified_cwes)}")233# pprint.pprint(unified_cwes)234 235print("\nTop 10 CWEs by occurrence (across all severities):")236for rank, (cwe, count) in enumerate(cwe_counter.most_common(10), start=1):237 print(f"{rank:2}. {cwe}: {count}")238 239print(f"\nSeverity types: {severity_types}")240print(f"CWEs divided by severity:")241pprint.pprint(cwes_counts_by_severity)242 243 244 