Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Read-only validation for the processed chatbot dataset (.csv or .pkl).4 5Discovers a dataset under backend/data/ or backend/artifacts/ and prints checks.6Does not modify any files.7"""8 9from __future__ import annotations10 11import sys12from pathlib import Path13 14import joblib15import pandas as pd16 17_BACKEND_DIR = Path(__file__).resolve().parent.parent18if str(_BACKEND_DIR) not in sys.path:19 sys.path.insert(0, str(_BACKEND_DIR))20 21from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT22 23REQUIRED_COLUMNS = [24 "QId",25 "Question",26 "AId",27 "Answer",28 "Score",29 "latest_score",30 "alternate",31 "date",32]33 34TEXT_TRUNCATE = 15035SAMPLE_N = 536NULL_WARN_FRACTION = 0.05 # warn if more than 5% of rows are null in a column37 38 39def _is_vector_artifact(path: Path) -> bool:40 """Exclude TF-IDF / vectorizer pickles from dataset auto-pick."""41 low = path.name.lower()42 if low == "vectorizer.pkl" or low == "question_vectors.pkl":43 return True44 if "vectorizer" in low:45 return True46 if "question_vector" in low or "question_vectors" in low:47 return True48 return False49 50 51def discover_dataset_path() -> Path:52 """53 Prefer chatbot_df.pkl, then final_chatbot_data.csv, then other csv/pkl in data/artifacts.54 """55 preferred = [56 ARTIFACTS_DIR / "chatbot_df.pkl",57 DATA_DIR / "final_chatbot_data.csv",58 ]59 for p in preferred:60 if p.is_file():61 return p62 63 if DATA_DIR.is_dir():64 csvs = sorted(DATA_DIR.glob("*.csv"))65 if csvs:66 return csvs[0]67 68 if ARTIFACTS_DIR.is_dir():69 pkls = sorted(ARTIFACTS_DIR.glob("*.pkl"))70 for p in pkls:71 if _is_vector_artifact(p):72 continue73 return p74 75 raise FileNotFoundError(76 "No dataset file found. Searched (paths resolved from this script, not cwd):\n"77 f" data dir: {DATA_DIR} (exists: {DATA_DIR.is_dir()})\n"78 f" artifacts dir: {ARTIFACTS_DIR} (exists: {ARTIFACTS_DIR.is_dir()})\n"79 "Prefer:\n"80 f" {ARTIFACTS_DIR / 'chatbot_df.pkl'}\n"81 f" {DATA_DIR / 'final_chatbot_data.csv'}\n"82 "or another .csv under backend/data/ or dataset .pkl under backend/artifacts/ "83 "(excluding vectorizer / question_vectors)."84 )85 86 87def load_dataset(path: Path) -> pd.DataFrame:88 suf = path.suffix.lower()89 if suf == ".pkl":90 obj = joblib.load(path)91 if not isinstance(obj, pd.DataFrame):92 raise TypeError(93 f"Expected pandas.DataFrame from {path}, got {type(obj).__name__}"94 )95 return obj96 if suf == ".csv":97 return pd.read_csv(path)98 raise ValueError(f"Unsupported extension: {path}")99 100 101def _memory_usage_mb(df: pd.DataFrame) -> float:102 return float(df.memory_usage(deep=True).sum()) / (1024 * 1024)103 104 105def _truncate(val: object, width: int = TEXT_TRUNCATE) -> str:106 s = str(val)107 if len(s) <= width:108 return s109 return s[: width - 3] + "..."110 111 112def _empty_text_mask(series: pd.Series) -> pd.Series:113 """True where value is non-null but empty/whitespace string."""114 return series.notna() & series.map(115 lambda x: isinstance(x, str) and len(x.strip()) == 0116 )117 118 119def main() -> int:120 print(f"This script: {Path(__file__).resolve()}")121 print(f"Repo root: {_REPO_ROOT}")122 print(f"Data dir: {DATA_DIR}")123 print(f"Artifacts: {ARTIFACTS_DIR}\n")124 125 try:126 path = discover_dataset_path()127 except FileNotFoundError as exc:128 print(f"ERROR: {exc}", file=sys.stderr)129 return 1130 131 print(f"Detected dataset file: {path}\n")132 133 try:134 df = load_dataset(path)135 except (OSError, TypeError, ValueError, UnicodeDecodeError) as exc:136 print(f"ERROR: Failed to load dataset: {exc}", file=sys.stderr)137 return 1138 139 warnings: list[str] = []140 141 # 1. Dataset info142 print("=== Dataset info ===")143 print(f" Shape: {df.shape[0]:,} rows × {df.shape[1]} columns")144 print(f" Columns ({len(df.columns)}): {list(df.columns)}")145 print(f" Memory usage (deep): {_memory_usage_mb(df):.4f} MB")146 print()147 148 # 2. Required columns149 print("=== Required columns ===")150 missing = [c for c in REQUIRED_COLUMNS if c not in df.columns]151 if missing:152 msg = f"Missing required columns: {missing}"153 print(f" FAIL: {msg}")154 warnings.append(msg)155 else:156 print(" OK: all required columns present")157 print()158 159 if missing:160 print("=== Warnings ===")161 for w in warnings:162 print(f" - {w}")163 return 1164 165 subset = df[REQUIRED_COLUMNS]166 167 # 3. Nulls168 print("=== Null values (per column) ===")169 null_counts = subset.isna().sum()170 for col in REQUIRED_COLUMNS:171 n = int(null_counts[col])172 frac = n / len(df) if len(df) else 0.0173 print(f" {col}: {n:,} ({100.0 * frac:.4f}%)")174 if frac > NULL_WARN_FRACTION:175 warnings.append(176 f"Too many nulls in '{col}': {n:,} ({100.0 * frac:.2f}% > {100 * NULL_WARN_FRACTION:.0f}%)"177 )178 print()179 180 # Empty strings in Question / Answer181 print("=== Empty / whitespace-only text ===")182 eq = _empty_text_mask(df["Question"])183 ea = _empty_text_mask(df["Answer"])184 n_eq = int(eq.sum())185 n_ea = int(ea.sum())186 print(f" Question (non-null but empty after strip): {n_eq:,}")187 print(f" Answer (non-null but empty after strip): {n_ea:,}")188 if n_eq:189 warnings.append(f"Empty Question text in {n_eq:,} row(s)")190 if n_ea:191 warnings.append(f"Empty Answer text in {n_ea:,} row(s)")192 print()193 194 # Duplicate QId195 print("=== Duplicate QId ===")196 dup_rows = df.duplicated(subset=["QId"], keep=False)197 n_dup_rows = int(dup_rows.sum())198 n_unique_qid = df["QId"].nunique()199 print(f" Unique QId values: {n_unique_qid:,}")200 print(f" Rows that share a QId with another row: {n_dup_rows:,}")201 if n_dup_rows:202 warnings.append(203 f"Duplicate QId: {n_dup_rows:,} rows appear in non-unique QId groups"204 )205 print()206 207 # 4. Samples208 print(f"=== Sample rows (first {SAMPLE_N}, Question/Answer truncated to {TEXT_TRUNCATE} chars) ===")209 sample = subset.head(SAMPLE_N)210 for idx, row in sample.iterrows():211 print(f" --- index {idx} ---")212 for col in REQUIRED_COLUMNS:213 if col in ("Question", "Answer"):214 print(f" {col}: {_truncate(row[col])}")215 else:216 print(f" {col}: {row[col]}")217 print()218 219 # 5. Warnings summary220 print("=== Warnings ===")221 if not warnings:222 print(" (none)")223 else:224 for w in warnings:225 print(f" - {w}")226 227 return 0228 229 230if __name__ == "__main__":231 raise SystemExit(main())232 