Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Read-only validation of processed chatbot data (pickle or CSV).4 5Does not modify data or retrieval logic.6"""7 8from __future__ import annotations9 10import sys11from pathlib import Path12 13import joblib14import pandas as pd15 16_BACKEND_DIR = Path(__file__).resolve().parent17if str(_BACKEND_DIR) not in sys.path:18 sys.path.insert(0, str(_BACKEND_DIR))19 20from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT21 22CHATBOT_DF_PKL = ARTIFACTS_DIR / "chatbot_df.pkl"23FINAL_CSV = DATA_DIR / "final_chatbot_data.csv"24 25REQUIRED_COLUMNS = [26 "QId",27 "Question",28 "AId",29 "Answer",30 "Score",31 "latest_score",32 "alternate",33 "date",34]35 36TEXT_TRUNCATE = 12037SAMPLE_N = 538 39 40def _load_dataframe() -> tuple[pd.DataFrame, str]:41 """Load chatbot_df.pkl if present, else final CSV."""42 if CHATBOT_DF_PKL.is_file():43 df = joblib.load(CHATBOT_DF_PKL)44 if not isinstance(df, pd.DataFrame):45 raise TypeError(f"Expected DataFrame from {CHATBOT_DF_PKL}, got {type(df).__name__}")46 return df, str(CHATBOT_DF_PKL)47 48 if FINAL_CSV.is_file():49 df = pd.read_csv(FINAL_CSV)50 return df, str(FINAL_CSV)51 52 raise FileNotFoundError(53 f"Neither artifact nor CSV found:\n {CHATBOT_DF_PKL}\n {FINAL_CSV}"54 )55 56 57def _truncate(val: object, width: int = TEXT_TRUNCATE) -> str:58 s = str(val)59 if len(s) <= width:60 return s61 return s[: width - 3] + "..."62 63 64def _non_empty_string_series(s: pd.Series) -> pd.Series:65 """True where value is a non-empty string (after strip); False for null or non-string."""66 def ok(v: object) -> bool:67 if pd.isna(v):68 return False69 if isinstance(v, str):70 return len(v.strip()) > 071 return False72 73 return s.map(ok)74 75 76def main() -> int:77 print("Processed chatbot data validation")78 print(f"Repo root: {REPO_ROOT}")79 80 try:81 df, source = _load_dataframe()82 except FileNotFoundError as exc:83 print(f"ERROR: {exc}", file=sys.stderr)84 return 185 except (OSError, TypeError, ValueError) as exc:86 print(f"ERROR: Failed to load data: {exc}", file=sys.stderr)87 return 188 89 print(f"Source: {source}")90 print(f"Shape: {df.shape[0]:,} rows × {df.shape[1]} columns")91 print(f"Column names ({len(df.columns)}): {list(df.columns)}")92 93 missing = [c for c in REQUIRED_COLUMNS if c not in df.columns]94 if missing:95 print(f"ERROR: Missing required columns: {missing}", file=sys.stderr)96 return 197 print("Required columns: OK (all present)")98 99 subset = df[REQUIRED_COLUMNS]100 null_counts = subset.isna().sum()101 print("\nNull counts (required columns):")102 for col in REQUIRED_COLUMNS:103 print(f" {col}: {int(null_counts[col]):,}")104 105 print(f"\nSample ({SAMPLE_N} rows, text truncated to {TEXT_TRUNCATE} chars):")106 sample = subset.head(SAMPLE_N)107 for idx, row in sample.iterrows():108 print(f" --- row index {idx} ---")109 for col in REQUIRED_COLUMNS:110 val = row[col]111 if col in ("Question", "Answer"):112 print(f" {col}: {_truncate(val)}")113 else:114 print(f" {col}: {val}")115 116 bad_q = ~_non_empty_string_series(df["Question"])117 bad_a = ~_non_empty_string_series(df["Answer"])118 n_bad_q = int(bad_q.sum())119 n_bad_a = int(bad_a.sum())120 121 if n_bad_q or n_bad_a:122 print(123 f"\nERROR: Question/Answer must be non-empty strings. "124 f"Bad Question rows: {n_bad_q:,}; bad Answer rows: {n_bad_a:,}",125 file=sys.stderr,126 )127 if n_bad_q:128 q_idx = df.index[bad_q][:5].tolist()129 print(f" First bad Question indices (up to 5): {q_idx}", file=sys.stderr)130 if n_bad_a:131 a_idx = df.index[bad_a][:5].tolist()132 print(f" First bad Answer indices (up to 5): {a_idx}", file=sys.stderr)133 return 1134 135 print("\nValidation: Question and Answer are non-empty strings for all rows (OK)")136 137 return 0138 139 140if __name__ == "__main__":141 raise SystemExit(main())142 