Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
test_processed_data.py142 linesDownload Raw Back to backend
1#!/usr/bin/env python32"""3Read-only validation of processed chatbot data (pickle or CSV).4 5Does not modify data or retrieval logic.6"""7 8from __future__ import annotations9 10import sys11from pathlib import Path12 13import joblib14import pandas as pd15 16_BACKEND_DIR = Path(__file__).resolve().parent17if str(_BACKEND_DIR) not in sys.path:18    sys.path.insert(0, str(_BACKEND_DIR))19 20from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT21 22CHATBOT_DF_PKL = ARTIFACTS_DIR / "chatbot_df.pkl"23FINAL_CSV = DATA_DIR / "final_chatbot_data.csv"24 25REQUIRED_COLUMNS = [26    "QId",27    "Question",28    "AId",29    "Answer",30    "Score",31    "latest_score",32    "alternate",33    "date",34]35 36TEXT_TRUNCATE = 12037SAMPLE_N = 538 39 40def _load_dataframe() -> tuple[pd.DataFrame, str]:41    """Load chatbot_df.pkl if present, else final CSV."""42    if CHATBOT_DF_PKL.is_file():43        df = joblib.load(CHATBOT_DF_PKL)44        if not isinstance(df, pd.DataFrame):45            raise TypeError(f"Expected DataFrame from {CHATBOT_DF_PKL}, got {type(df).__name__}")46        return df, str(CHATBOT_DF_PKL)47 48    if FINAL_CSV.is_file():49        df = pd.read_csv(FINAL_CSV)50        return df, str(FINAL_CSV)51 52    raise FileNotFoundError(53        f"Neither artifact nor CSV found:\n  {CHATBOT_DF_PKL}\n  {FINAL_CSV}"54    )55 56 57def _truncate(val: object, width: int = TEXT_TRUNCATE) -> str:58    s = str(val)59    if len(s) <= width:60        return s61    return s[: width - 3] + "..."62 63 64def _non_empty_string_series(s: pd.Series) -> pd.Series:65    """True where value is a non-empty string (after strip); False for null or non-string."""66    def ok(v: object) -> bool:67        if pd.isna(v):68            return False69        if isinstance(v, str):70            return len(v.strip()) > 071        return False72 73    return s.map(ok)74 75 76def main() -> int:77    print("Processed chatbot data validation")78    print(f"Repo root: {REPO_ROOT}")79 80    try:81        df, source = _load_dataframe()82    except FileNotFoundError as exc:83        print(f"ERROR: {exc}", file=sys.stderr)84        return 185    except (OSError, TypeError, ValueError) as exc:86        print(f"ERROR: Failed to load data: {exc}", file=sys.stderr)87        return 188 89    print(f"Source: {source}")90    print(f"Shape: {df.shape[0]:,} rows × {df.shape[1]} columns")91    print(f"Column names ({len(df.columns)}): {list(df.columns)}")92 93    missing = [c for c in REQUIRED_COLUMNS if c not in df.columns]94    if missing:95        print(f"ERROR: Missing required columns: {missing}", file=sys.stderr)96        return 197    print("Required columns: OK (all present)")98 99    subset = df[REQUIRED_COLUMNS]100    null_counts = subset.isna().sum()101    print("\nNull counts (required columns):")102    for col in REQUIRED_COLUMNS:103        print(f"  {col}: {int(null_counts[col]):,}")104 105    print(f"\nSample ({SAMPLE_N} rows, text truncated to {TEXT_TRUNCATE} chars):")106    sample = subset.head(SAMPLE_N)107    for idx, row in sample.iterrows():108        print(f"  --- row index {idx} ---")109        for col in REQUIRED_COLUMNS:110            val = row[col]111            if col in ("Question", "Answer"):112                print(f"    {col}: {_truncate(val)}")113            else:114                print(f"    {col}: {val}")115 116    bad_q = ~_non_empty_string_series(df["Question"])117    bad_a = ~_non_empty_string_series(df["Answer"])118    n_bad_q = int(bad_q.sum())119    n_bad_a = int(bad_a.sum())120 121    if n_bad_q or n_bad_a:122        print(123            f"\nERROR: Question/Answer must be non-empty strings. "124            f"Bad Question rows: {n_bad_q:,}; bad Answer rows: {n_bad_a:,}",125            file=sys.stderr,126        )127        if n_bad_q:128            q_idx = df.index[bad_q][:5].tolist()129            print(f"  First bad Question indices (up to 5): {q_idx}", file=sys.stderr)130        if n_bad_a:131            a_idx = df.index[bad_a][:5].tolist()132            print(f"  First bad Answer indices (up to 5): {a_idx}", file=sys.stderr)133        return 1134 135    print("\nValidation: Question and Answer are non-empty strings for all rows (OK)")136 137    return 0138 139 140if __name__ == "__main__":141    raise SystemExit(main())142