Team Ai
Apppublic

OSS-forge/CodeQualityEval

sourceHugging Facecc-by-sa-4.0updated 11mo agoView on Hugging Face
12likes
process_semgrep_results_python.py244 linesDownload Raw Back to 4_Code_Security_Analysis
1import json2import pprint3import argparse4import re5from collections import defaultdict6from collections import defaultdict, Counter 7 8"""9Read filename and max batch number from commandline. Rename all the files to have a single name and number.10"""11 12parser = argparse.ArgumentParser(description='Process Semgrep results.')13parser.add_argument('json_filename', type=str, help='Base filename for Semgrep JSON results')14parser.add_argument('max_batch_num', type=int, help='Maximum batch number to process')15 16args = parser.parse_args()17json_filename = args.json_filename18max_batch_num = args.max_batch_num19 20"""21Read json file in batches and create a single list of total errors, results, and scanned files.22Count number of issues, number of scanned file, number of files that caused errors and compute issues percentage.23 24NB: skipped files contain errors (already accounted for in total_errors) and incompatible rules due to version 25and language (filtered out from the errors)26"""27 28total_errors = []29total_results = []30total_scanned = []31total_skipped = []32 33for i in range(1, max_batch_num + 1):34    json_filename_complete = f"{json_filename}_{i}.json"35    filtered_errors = []36    with open(json_filename_complete, 'r', encoding='utf-8') as results_f:37        samples = json.load(results_f)38        filtered_errors.extend(39            [error for error in samples['errors'] if not error['path'].startswith('https:/semgrep.dev/...')]40        )  # Filtering out incompatible rules41        total_errors.extend(filtered_errors)42        total_results.extend(samples['results'])43        total_scanned.extend(samples['paths']['scanned'])44        total_skipped.extend(samples['paths']['skipped'])45 46"""47Calculate file number from the filename to obtain the dataset line number and insert it into the path field.48This is done to filter out duplicates.49"""50pattern = r'.*_(\d+)\.py'51 52def calculate_line_number(filename):53    match = re.match(pattern, filename)54    # print(f"Filename: {filename}, Match: {int(match.group(1)) if match else None}")55    return int(match.group(1)) if match else None56 57 58for error in total_errors:59    error['path'] = calculate_line_number(error['path'])60 61for result in total_results:62    result['path'] = calculate_line_number(result['path'])63 64for i in range(len(total_scanned)):65    total_scanned[i] = calculate_line_number(total_scanned[i])66 67 68"""69Remove duplicates from the errors and results lists.70_______________________71dedup_err is the list of errors w/o duplicates72dedup_res is the list of defective functions (i.e., w/o duplicated issues)73total_results is the list of issues w/o errors74dedup_res_no_errors is the list of defective functions w/o errors75"""76 77dedup_err = {err['path'] for err in total_errors}78dedup_res = {res['path'] for res in total_results}79 80dedup_res_no_errors = [res for res in dedup_res if res not in dedup_err]81total_results = [res for res in total_results if res['path'] not in dedup_err]82 83"""84Normalize CWE names dynamically to ensure uniqueness.85"""86 87def extract_cwe_number(cwe_name):88    """Extract CWE-XXX format from any given CWE description."""89    match = re.match(r"(CWE-\d+)", cwe_name, re.IGNORECASE)90    return match.group(1) if match else cwe_name91 92 93"""94Divide issues based on category type. 95Since not all issues are correctly categories (i.e., missing "category" field), 96we select them based on whether they have a "CWE" field.97"""98 99security_issues = []100seen_issues = set()101severity_types = set()102normalized_cwe_dict = defaultdict(str)103 104# Process security issues and normalize CWEs105for result in total_results:106    metadata = result.get('extra', {}).get('metadata', {})107    cwes = metadata.get('cwe')108    severity = result.get('extra', {}).get('severity')109 110    if cwes:111        if isinstance(cwes, list):112            updated_cwes = []113            for cwe in cwes:114                base_cwe = extract_cwe_number(cwe)115                if base_cwe in normalized_cwe_dict:116                    standardized_cwe = max(normalized_cwe_dict[base_cwe], cwe, key=len)117                else:118                    standardized_cwe = cwe  # Keep first occurrence as reference119                    normalized_cwe_dict[base_cwe] = standardized_cwe120                updated_cwes.append(standardized_cwe)121            result['extra']['metadata']['cwe'] = [cwe.upper() for cwe in updated_cwes]122        else:123            cwes = f"{cwes.upper()}"124            base_cwe = extract_cwe_number(cwes)125            if base_cwe in normalized_cwe_dict:126                standardized_cwe = max(normalized_cwe_dict[base_cwe], cwes, key=len)127            else:128                standardized_cwe = cwes  # Keep first occurrence as reference129                normalized_cwe_dict[base_cwe] = standardized_cwe130            result['extra']['metadata']['cwe'] = standardized_cwe.upper()131 132        # Use a unique identifier for each issue (path, CWE, severity, and message)133        issue_id = (134            result['path'], 135            tuple(sorted(result['extra']['metadata']['cwe'])),  # Ensure consistent ordering of CWEs136            result['extra'].get('severity', ''), 137            result['extra'].get('lines', '').strip(),  # Remove accidental whitespace138        )139 140        if issue_id not in seen_issues:141            seen_issues.add(issue_id)  # Add to set to track unique issues142            security_issues.append(result)143 144        if severity:145            severity_types.add(severity)146 147# Deduplicate CWEs by keeping only the longest description for each CWE number148deduplicated_cwes = {}149 150for base_cwe, cwe_description in normalized_cwe_dict.items():151    base_cwe = base_cwe.upper()  # Ensure "CWE" is always uppercase152    cwe_description = cwe_description.strip()  # Remove any accidental spaces153 154    # Keep the longest description per CWE number155    if base_cwe not in deduplicated_cwes or len(cwe_description) > len(deduplicated_cwes[base_cwe]):156        deduplicated_cwes[base_cwe] = cwe_description157 158unified_cwes = set(deduplicated_cwes.values())159 160for result in security_issues:161    metadata = result.get('extra', {}).get('metadata', {})162    cwes = metadata.get('cwe')163 164    if cwes:165        if isinstance(cwes, list):166            result['extra']['metadata']['cwe'] = [deduplicated_cwes[extract_cwe_number(cwe).upper()] for cwe in cwes]167        else:168            result['extra']['metadata']['cwe'] = deduplicated_cwes[extract_cwe_number(cwes).upper()]169 170"""171NEW: Compute and print the Top‑10 most frequent CWEs across the dataset172"""173 174cwe_counter = Counter()175for issue in security_issues:176    cwes = issue['extra']['metadata']['cwe']177    if isinstance(cwes, list):178        cwe_counter.update(cwes)179    else:180        cwe_counter.update([cwes])181 182 183"""184Divide security-related issues by CWE severity category.185"""186 187cwes_by_severity = {severity: {} for severity in severity_types}188 189for issue in security_issues:190    metadata = issue.get('extra', {}).get('metadata', {})191    cwes = metadata.get('cwe')192    severity = issue.get('extra', {}).get('severity')193 194    if severity and cwes:195        if isinstance(cwes, list):196            for cwe in cwes:197                if cwe not in cwes_by_severity[severity]:198                    cwes_by_severity[severity][cwe] = []199                cwes_by_severity[severity][cwe].append(issue)200        else:201            if cwes not in cwes_by_severity[severity]:202                cwes_by_severity[severity][cwes] = []203            cwes_by_severity[severity][cwes].append(issue)204 205cwes_counts_by_severity = {206    severity: {cwe: len(issues) for cwe, issues in cwes_dict.items()}207    for severity, cwes_dict in cwes_by_severity.items()208}209 210"""211Compute percentages of defects, errors and clean functions.212 213NB: security_issues is already error-free because "total_results" is error free 214-> we only need to remove path duplicates to obtain the number of defective functions (only security)215"""216 217# Computing defective functions (i.e., removing duplicate security issues). 218# We only need the number and path to later remove them from the dataset219defective_func_security_set = {issue['path'] for issue in security_issues}220 221defective_func_rate = (len(defective_func_security_set) / len(total_scanned)) * 100222errors_rate = (len(dedup_err) / len(total_scanned)) * 100223clean_rate = ((len(total_scanned) - len(defective_func_security_set) - len(dedup_err)) / len(total_scanned)) * 100224 225print(f"Total skipped functions: {len(total_skipped)} (errors + incompatible rules)")226print(f"Total scanned functions: {len(total_scanned)} (100%)")227print(f"Total clean functions: {len(total_scanned)-len(defective_func_security_set)-len(dedup_err)} ({clean_rate:.2f}%)")228print(f"Total defective functions (excluding errors): {len(defective_func_security_set)} ({defective_func_rate:.2f}%)")229print(f"Total errors: {len(total_errors)}. Errors w/o duplicates: {len(dedup_err)} ({errors_rate:.2f}%)")230print(f"Total issues (considering multiple issues per function and excluding errors): {len(security_issues)}")231 232print(f"\nFinal Unified CWE Set (without duplicates): {len(unified_cwes)}")233# pprint.pprint(unified_cwes)234 235print("\nTop 10 CWEs by occurrence (across all severities):")236for rank, (cwe, count) in enumerate(cwe_counter.most_common(10), start=1):237    print(f"{rank:2}. {cwe}: {count}")238 239print(f"\nSeverity types: {severity_types}")240print(f"CWEs divided by severity:")241pprint.pprint(cwes_counts_by_severity)242 243 244