joshnavip/ai-code-detection
0
1import pandas as pd2 3# ===============================4# PATHS5# ===============================6INPUT_PATH = "dataset/raw/raw_dataset.csv"7OUTPUT_PATH = "dataset/processed/dataset_step1_length_normalized.csv"8 9# ===============================10# PARAMETERS11# ===============================12MIN_LINES = 413MAX_LINES = 10014 15# ===============================16# UTILITY FUNCTIONS17# ===============================18def get_clean_lines(code):19 if not isinstance(code, str):20 return []21 22 lines = code.splitlines()23 24 # Remove leading empty lines25 while lines and lines[0].strip() == "":26 lines.pop(0)27 28 # Remove trailing empty lines29 while lines and lines[-1].strip() == "":30 lines.pop()31 32 return lines33 34 35def get_truncation_marker(language):36 if language == "Python":37 return "# ... truncated ..."38 elif language == "Java":39 return "// ... truncated ..."40 else:41 return "... truncated ..."42 43 44def truncate_middle(lines, max_lines, language):45 if len(lines) <= max_lines:46 return lines47 48 head = lines[: max_lines // 2]49 tail = lines[-(max_lines // 2):]50 marker = get_truncation_marker(language)51 52 return head + [marker] + tail53 54 55def normalize_code(code, language):56 lines = get_clean_lines(code)57 58 if len(lines) < MIN_LINES:59 return None60 61 if len(lines) > MAX_LINES:62 lines = truncate_middle(lines, MAX_LINES, language)63 64 return "\n".join(lines)65 66# ===============================67# MAIN PIPELINE68# ===============================69def main():70 print("Loading raw dataset...")71 df = pd.read_csv(INPUT_PATH)72 print(f"Initial dataset size: {len(df)} rows")73 74 # -------------------------------75 # Column validation76 # -------------------------------77 required_cols = ["Code_Text","Language"]78 missing = [c for c in required_cols if c not in df.columns]79 if missing:80 raise ValueError(f"Missing required columns: {missing}")81 82 # -------------------------------83 # Normalize code length84 # -------------------------------85 print("Normalizing code length (language-aware)...")86 87 df["normalized_code"] = df.apply(88 lambda row: normalize_code(row["Code_Text"], row["Language"]),89 axis=190 )91 92 # Drop rows filtered out by normalization93 df = df[df["normalized_code"].notnull()].reset_index(drop=True)94 95 # -------------------------------96 # Metadata for analysis/debugging97 # -------------------------------98 df["original_line_count"] = df["Code_Text"].apply(99 lambda c: len(get_clean_lines(c))100 )101 102 df["normalized_line_count"] = df["normalized_code"].apply(103 lambda c: len(get_clean_lines(c))104 )105 106 print(f"Final dataset size after normalization: {len(df)} rows")107 108 # -------------------------------109 # Save output110 # -------------------------------111 print("Saving processed dataset...")112 df.to_csv(OUTPUT_PATH, index=False)113 114 print("Step 1 completed successfully ✅")115 print(f"Output file saved at: {OUTPUT_PATH}")116 117 118if __name__ == "__main__":119 main()120 