Team Ai
Apppublic

joshnavip/ai-code-detection

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
step1_length_normalization.py120 linesDownload Raw Back to preprocessing
1import pandas as pd2 3# ===============================4# PATHS5# ===============================6INPUT_PATH = "dataset/raw/raw_dataset.csv"7OUTPUT_PATH = "dataset/processed/dataset_step1_length_normalized.csv"8 9# ===============================10# PARAMETERS11# ===============================12MIN_LINES = 413MAX_LINES = 10014 15# ===============================16# UTILITY FUNCTIONS17# ===============================18def get_clean_lines(code):19    if not isinstance(code, str):20        return []21 22    lines = code.splitlines()23 24    # Remove leading empty lines25    while lines and lines[0].strip() == "":26        lines.pop(0)27 28    # Remove trailing empty lines29    while lines and lines[-1].strip() == "":30        lines.pop()31 32    return lines33 34 35def get_truncation_marker(language):36    if language == "Python":37        return "# ... truncated ..."38    elif language == "Java":39        return "// ... truncated ..."40    else:41        return "... truncated ..."42 43 44def truncate_middle(lines, max_lines, language):45    if len(lines) <= max_lines:46        return lines47 48    head = lines[: max_lines // 2]49    tail = lines[-(max_lines // 2):]50    marker = get_truncation_marker(language)51 52    return head + [marker] + tail53 54 55def normalize_code(code, language):56    lines = get_clean_lines(code)57 58    if len(lines) < MIN_LINES:59        return None60 61    if len(lines) > MAX_LINES:62        lines = truncate_middle(lines, MAX_LINES, language)63 64    return "\n".join(lines)65 66# ===============================67# MAIN PIPELINE68# ===============================69def main():70    print("Loading raw dataset...")71    df = pd.read_csv(INPUT_PATH)72    print(f"Initial dataset size: {len(df)} rows")73 74    # -------------------------------75    # Column validation76    # -------------------------------77    required_cols = ["Code_Text","Language"]78    missing = [c for c in required_cols if c not in df.columns]79    if missing:80        raise ValueError(f"Missing required columns: {missing}")81 82    # -------------------------------83    # Normalize code length84    # -------------------------------85    print("Normalizing code length (language-aware)...")86 87    df["normalized_code"] = df.apply(88        lambda row: normalize_code(row["Code_Text"], row["Language"]),89        axis=190    )91 92    # Drop rows filtered out by normalization93    df = df[df["normalized_code"].notnull()].reset_index(drop=True)94 95    # -------------------------------96    # Metadata for analysis/debugging97    # -------------------------------98    df["original_line_count"] = df["Code_Text"].apply(99        lambda c: len(get_clean_lines(c))100    )101 102    df["normalized_line_count"] = df["normalized_code"].apply(103        lambda c: len(get_clean_lines(c))104    )105 106    print(f"Final dataset size after normalization: {len(df)} rows")107 108    # -------------------------------109    # Save output110    # -------------------------------111    print("Saving processed dataset...")112    df.to_csv(OUTPUT_PATH, index=False)113 114    print("Step 1 completed successfully ✅")115    print(f"Output file saved at: {OUTPUT_PATH}")116 117 118if __name__ == "__main__":119    main()120