Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
verify_project.py189 linesDownload Raw Back to debug
1#!/usr/bin/env python32"""3Full project verification (read-only): folders, expected files, dataset/vectors4consistency, one retrieval smoke test. Does not modify data or application logic.5 6Run:7  python backend/debug/verify_project.py8"""9 10from __future__ import annotations11 12import sys13import traceback14from pathlib import Path15 16import joblib17import numpy as np18import pandas as pd19from bs4 import BeautifulSoup20from sklearn.metrics.pairwise import cosine_similarity21 22_BACKEND_DIR = Path(__file__).resolve().parent.parent23if str(_BACKEND_DIR) not in sys.path:24    sys.path.insert(0, str(_BACKEND_DIR))25 26from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT27 28from app.services.retriever import getAnswerWithHighestScore  # noqa: E40229 30REQUIRED_DATASET_COLUMNS = [31    "QId",32    "Question",33    "AId",34    "Answer",35    "Score",36    "latest_score",37    "alternate",38    "date",39]40 41EXPECTED_FILES: list[tuple[str, Path]] = [42    ("AnswersV2.csv", DATA_DIR / "AnswersV2.csv"),43    ("final_chatbot_data.csv", DATA_DIR / "final_chatbot_data.csv"),44    ("vectorizer.pkl", ARTIFACTS_DIR / "vectorizer.pkl"),45    ("question_vectors.pkl", ARTIFACTS_DIR / "question_vectors.pkl"),46    ("chatbot_df.pkl", ARTIFACTS_DIR / "chatbot_df.pkl"),47]48 49TEST_QUERY = "What is list comprehension in Python?"50 51 52def _size_mb(path: Path) -> float:53    return path.stat().st_size / (1024 * 1024)54 55 56def _print_file_line(label: str, path: Path) -> bool:57    if not path.is_file():58        print(f"  {label}: MISSING (expected {path})")59        return False60    try:61        mb = _size_mb(path)62    except OSError as exc:63        print(f"  {label}: ERROR stat {path}: {exc}")64        return False65    print(f"  {label}: OK")66    print(f"    path: {path}")67    print(f"    size: {mb:.4f} MB")68    return True69 70 71def main() -> int:72    print("=== Project verification (paths resolved from this script) ===\n")73    print(f"This script: {Path(__file__).resolve()}")74    print(f"Repo root:   {REPO_ROOT}")75    print(f"Data dir:    {DATA_DIR}")76    print(f"Artifacts:   {ARTIFACTS_DIR}\n")77 78    issues: list[str] = []79 80    # 1. Folders81    print("--- Folders ---")82    for label, p in [("backend/data/", DATA_DIR), ("backend/artifacts/", ARTIFACTS_DIR)]:83        if p.is_dir():84            print(f"  {label} OK: {p}")85        else:86            print(f"  {label} MISSING or not a directory: {p}")87            issues.append(f"folder missing: {p}")88    print()89 90    # 2–3. Expected files91    print("--- Expected files ---")92    all_files_ok = True93    for name, path in EXPECTED_FILES:94        if not _print_file_line(name, path):95            all_files_ok = False96            issues.append(f"missing file: {name}")97        print()98    if not all_files_ok:99        print("--- Backend ready for React frontend integration: NO ---")100        print("Reason: one or more expected files are missing.")101        for i in issues:102            print(f"  - {i}")103        return 1104 105    # 4. Dataset columns, row counts, vector alignment (use same artifacts as FastAPI)106    print("--- Dataset & vectors validation ---")107    df_path = ARTIFACTS_DIR / "chatbot_df.pkl"108    v_path = ARTIFACTS_DIR / "vectorizer.pkl"109    qv_path = ARTIFACTS_DIR / "question_vectors.pkl"110 111    try:112        df = joblib.load(df_path)113        if not isinstance(df, pd.DataFrame):114            raise TypeError(f"chatbot_df.pkl must be DataFrame, got {type(df).__name__}")115        Question_vectors = joblib.load(qv_path)116    except Exception as exc:117        print(f"ERROR: failed to load dataframe or vectors: {exc}", file=sys.stderr)118        traceback.print_exc()119        print("\n--- Backend ready for React frontend integration: NO ---")120        return 1121 122    missing_cols = [c for c in REQUIRED_DATASET_COLUMNS if c not in df.columns]123    if missing_cols:124        print(f"  FAIL: missing columns: {missing_cols}")125        issues.append(f"missing columns: {missing_cols}")126    else:127        print(f"  OK: required columns present ({len(REQUIRED_DATASET_COLUMNS)} fields)")128 129    n_rows = len(df)130    print(f"  Dataframe row count: {n_rows:,}")131 132    try:133        n_vec = int(Question_vectors.shape[0])134    except Exception as exc:135        print(f"  FAIL: question_vectors shape: {exc}")136        issues.append("invalid question_vectors")137        n_vec = -1138 139    if n_vec >= 0:140        print(f"  Question_vectors rows: {n_vec:,}")141        if n_vec == n_rows:142            print("  OK: vector row count matches dataframe row count")143        else:144            print(145                f"  FAIL: row mismatch (dataframe {n_rows:,} vs vectors {n_vec:,})"146            )147            issues.append("vector rows != dataframe rows")148 149    if missing_cols or n_vec != n_rows or n_vec < 0:150        print("\n--- Backend ready for React frontend integration: NO ---")151        for i in issues:152            print(f"  - {i}")153        return 1154 155    print()156 157    # 5. One test query (same pipeline as retriever / test_artifacts)158    print(f"--- Test query ---")159    print(f"  Query: {TEST_QUERY!r}\n")160    try:161        vectorizer = joblib.load(v_path)162        input_question = BeautifulSoup(TEST_QUERY).get_text()163        input_question_vector = vectorizer.transform([input_question])164        similarities = cosine_similarity(input_question_vector, Question_vectors)165        closest = np.argmax(similarities, axis=1)166        row = int(closest[0])167        qid = df.iloc[row, 0]168        answers = df.loc[df["QId"] == qid]169        triple = getAnswerWithHighestScore(answers, df)170        print(f"  OK: inference completed (QId={int(qid)}, AId={int(triple[1])})")171        print(f"  Answer preview: {str(triple[0])[:200]}{'...' if len(str(triple[0])) > 200 else ''}")172    except Exception as exc:173        print(f"  FAIL: {exc}", file=sys.stderr)174        traceback.print_exc()175        print("\n--- Backend ready for React frontend integration: NO ---")176        print("Reason: test query failed.")177        return 1178 179    print()180    print("--- Backend ready for React frontend integration: YES ---")181    print("  All expected artifacts present, dataset validates, vectors align,")182    print("  and a smoke retrieval query succeeded. Start the API with uvicorn")183    print("  (e.g. from backend/: uvicorn app.main:app --host 127.0.0.1 --port 8000).")184    return 0185 186 187if __name__ == "__main__":188    raise SystemExit(main())189