Team Ai
Modelpublic

Harshtech1/CARZero-Replication

sourceHugging Faceupdated 4mo agoView on Hugging Face
0likes
preprocess_reports.py40 linesDownload Raw Back to data
1import pandas as pd2import re3 4def mock_llm_disease_extractor(raw_report):5    """6    This simulates the LLM prompt: 7    'Extract the primary disease from this report and format it as: There is [disease].'8    """9    report = str(raw_report).lower()10    11    # Simulate LLM extracting conditions12    if "cardiomegaly" in report or "heart is enlarged" in report:13        return "There is cardiomegaly."14    elif "effusion" in report:15        return "There is pleural effusion."16    elif "opacity" in report or "consolidation" in report:17        return "There is opacity."18    elif "pneumothorax" in report:19        return "There is pneumothorax."20    elif "normal" in report or "clear" in report:21        return "There is no disease."22    else:23        return "There is an unspecified abnormality."24 25def main():26    print("๐Ÿ“‚ Loading raw Open-I medical reports...")27    # The CSV downloaded from Kaggle28    df = pd.read_csv('indiana_reports.csv')29    30    print("๐Ÿง  Passing reports through LLM formatting pipeline...")31    # Apply our extractor to the 'findings' column32    df['llm_cleaned_prompt'] = df['findings'].apply(mock_llm_disease_extractor)33    34    # Save the new dataset specifically for CARZero35    output_file = 'carzero_cleaned_reports.csv'36    df.to_csv(output_file, index=False)37    print(f"โœ… Successfully generated unified LLM dataset: {output_file}")38 39if __name__ == "__main__":40    main()