Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
validate_dataset.py232 linesDownload Raw Back to debug
1#!/usr/bin/env python32"""3Read-only validation for the processed chatbot dataset (.csv or .pkl).4 5Discovers a dataset under backend/data/ or backend/artifacts/ and prints checks.6Does not modify any files.7"""8 9from __future__ import annotations10 11import sys12from pathlib import Path13 14import joblib15import pandas as pd16 17_BACKEND_DIR = Path(__file__).resolve().parent.parent18if str(_BACKEND_DIR) not in sys.path:19    sys.path.insert(0, str(_BACKEND_DIR))20 21from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT22 23REQUIRED_COLUMNS = [24    "QId",25    "Question",26    "AId",27    "Answer",28    "Score",29    "latest_score",30    "alternate",31    "date",32]33 34TEXT_TRUNCATE = 15035SAMPLE_N = 536NULL_WARN_FRACTION = 0.05  # warn if more than 5% of rows are null in a column37 38 39def _is_vector_artifact(path: Path) -> bool:40    """Exclude TF-IDF / vectorizer pickles from dataset auto-pick."""41    low = path.name.lower()42    if low == "vectorizer.pkl" or low == "question_vectors.pkl":43        return True44    if "vectorizer" in low:45        return True46    if "question_vector" in low or "question_vectors" in low:47        return True48    return False49 50 51def discover_dataset_path() -> Path:52    """53    Prefer chatbot_df.pkl, then final_chatbot_data.csv, then other csv/pkl in data/artifacts.54    """55    preferred = [56        ARTIFACTS_DIR / "chatbot_df.pkl",57        DATA_DIR / "final_chatbot_data.csv",58    ]59    for p in preferred:60        if p.is_file():61            return p62 63    if DATA_DIR.is_dir():64        csvs = sorted(DATA_DIR.glob("*.csv"))65        if csvs:66            return csvs[0]67 68    if ARTIFACTS_DIR.is_dir():69        pkls = sorted(ARTIFACTS_DIR.glob("*.pkl"))70        for p in pkls:71            if _is_vector_artifact(p):72                continue73            return p74 75    raise FileNotFoundError(76        "No dataset file found. Searched (paths resolved from this script, not cwd):\n"77        f"  data dir:      {DATA_DIR} (exists: {DATA_DIR.is_dir()})\n"78        f"  artifacts dir: {ARTIFACTS_DIR} (exists: {ARTIFACTS_DIR.is_dir()})\n"79        "Prefer:\n"80        f"  {ARTIFACTS_DIR / 'chatbot_df.pkl'}\n"81        f"  {DATA_DIR / 'final_chatbot_data.csv'}\n"82        "or another .csv under backend/data/ or dataset .pkl under backend/artifacts/ "83        "(excluding vectorizer / question_vectors)."84    )85 86 87def load_dataset(path: Path) -> pd.DataFrame:88    suf = path.suffix.lower()89    if suf == ".pkl":90        obj = joblib.load(path)91        if not isinstance(obj, pd.DataFrame):92            raise TypeError(93                f"Expected pandas.DataFrame from {path}, got {type(obj).__name__}"94            )95        return obj96    if suf == ".csv":97        return pd.read_csv(path)98    raise ValueError(f"Unsupported extension: {path}")99 100 101def _memory_usage_mb(df: pd.DataFrame) -> float:102    return float(df.memory_usage(deep=True).sum()) / (1024 * 1024)103 104 105def _truncate(val: object, width: int = TEXT_TRUNCATE) -> str:106    s = str(val)107    if len(s) <= width:108        return s109    return s[: width - 3] + "..."110 111 112def _empty_text_mask(series: pd.Series) -> pd.Series:113    """True where value is non-null but empty/whitespace string."""114    return series.notna() & series.map(115        lambda x: isinstance(x, str) and len(x.strip()) == 0116    )117 118 119def main() -> int:120    print(f"This script: {Path(__file__).resolve()}")121    print(f"Repo root:   {_REPO_ROOT}")122    print(f"Data dir:    {DATA_DIR}")123    print(f"Artifacts:   {ARTIFACTS_DIR}\n")124 125    try:126        path = discover_dataset_path()127    except FileNotFoundError as exc:128        print(f"ERROR: {exc}", file=sys.stderr)129        return 1130 131    print(f"Detected dataset file: {path}\n")132 133    try:134        df = load_dataset(path)135    except (OSError, TypeError, ValueError, UnicodeDecodeError) as exc:136        print(f"ERROR: Failed to load dataset: {exc}", file=sys.stderr)137        return 1138 139    warnings: list[str] = []140 141    # 1. Dataset info142    print("=== Dataset info ===")143    print(f"  Shape: {df.shape[0]:,} rows × {df.shape[1]} columns")144    print(f"  Columns ({len(df.columns)}): {list(df.columns)}")145    print(f"  Memory usage (deep): {_memory_usage_mb(df):.4f} MB")146    print()147 148    # 2. Required columns149    print("=== Required columns ===")150    missing = [c for c in REQUIRED_COLUMNS if c not in df.columns]151    if missing:152        msg = f"Missing required columns: {missing}"153        print(f"  FAIL: {msg}")154        warnings.append(msg)155    else:156        print("  OK: all required columns present")157    print()158 159    if missing:160        print("=== Warnings ===")161        for w in warnings:162            print(f"  - {w}")163        return 1164 165    subset = df[REQUIRED_COLUMNS]166 167    # 3. Nulls168    print("=== Null values (per column) ===")169    null_counts = subset.isna().sum()170    for col in REQUIRED_COLUMNS:171        n = int(null_counts[col])172        frac = n / len(df) if len(df) else 0.0173        print(f"  {col}: {n:,} ({100.0 * frac:.4f}%)")174        if frac > NULL_WARN_FRACTION:175            warnings.append(176                f"Too many nulls in '{col}': {n:,} ({100.0 * frac:.2f}% > {100 * NULL_WARN_FRACTION:.0f}%)"177            )178    print()179 180    # Empty strings in Question / Answer181    print("=== Empty / whitespace-only text ===")182    eq = _empty_text_mask(df["Question"])183    ea = _empty_text_mask(df["Answer"])184    n_eq = int(eq.sum())185    n_ea = int(ea.sum())186    print(f"  Question (non-null but empty after strip): {n_eq:,}")187    print(f"  Answer (non-null but empty after strip): {n_ea:,}")188    if n_eq:189        warnings.append(f"Empty Question text in {n_eq:,} row(s)")190    if n_ea:191        warnings.append(f"Empty Answer text in {n_ea:,} row(s)")192    print()193 194    # Duplicate QId195    print("=== Duplicate QId ===")196    dup_rows = df.duplicated(subset=["QId"], keep=False)197    n_dup_rows = int(dup_rows.sum())198    n_unique_qid = df["QId"].nunique()199    print(f"  Unique QId values: {n_unique_qid:,}")200    print(f"  Rows that share a QId with another row: {n_dup_rows:,}")201    if n_dup_rows:202        warnings.append(203            f"Duplicate QId: {n_dup_rows:,} rows appear in non-unique QId groups"204        )205    print()206 207    # 4. Samples208    print(f"=== Sample rows (first {SAMPLE_N}, Question/Answer truncated to {TEXT_TRUNCATE} chars) ===")209    sample = subset.head(SAMPLE_N)210    for idx, row in sample.iterrows():211        print(f"  --- index {idx} ---")212        for col in REQUIRED_COLUMNS:213            if col in ("Question", "Answer"):214                print(f"    {col}: {_truncate(row[col])}")215            else:216                print(f"    {col}: {row[col]}")217        print()218 219    # 5. Warnings summary220    print("=== Warnings ===")221    if not warnings:222        print("  (none)")223    else:224        for w in warnings:225            print(f"  - {w}")226 227    return 0228 229 230if __name__ == "__main__":231    raise SystemExit(main())232