Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Full project verification (read-only): folders, expected files, dataset/vectors4consistency, one retrieval smoke test. Does not modify data or application logic.5 6Run:7 python backend/debug/verify_project.py8"""9 10from __future__ import annotations11 12import sys13import traceback14from pathlib import Path15 16import joblib17import numpy as np18import pandas as pd19from bs4 import BeautifulSoup20from sklearn.metrics.pairwise import cosine_similarity21 22_BACKEND_DIR = Path(__file__).resolve().parent.parent23if str(_BACKEND_DIR) not in sys.path:24 sys.path.insert(0, str(_BACKEND_DIR))25 26from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT27 28from app.services.retriever import getAnswerWithHighestScore # noqa: E40229 30REQUIRED_DATASET_COLUMNS = [31 "QId",32 "Question",33 "AId",34 "Answer",35 "Score",36 "latest_score",37 "alternate",38 "date",39]40 41EXPECTED_FILES: list[tuple[str, Path]] = [42 ("AnswersV2.csv", DATA_DIR / "AnswersV2.csv"),43 ("final_chatbot_data.csv", DATA_DIR / "final_chatbot_data.csv"),44 ("vectorizer.pkl", ARTIFACTS_DIR / "vectorizer.pkl"),45 ("question_vectors.pkl", ARTIFACTS_DIR / "question_vectors.pkl"),46 ("chatbot_df.pkl", ARTIFACTS_DIR / "chatbot_df.pkl"),47]48 49TEST_QUERY = "What is list comprehension in Python?"50 51 52def _size_mb(path: Path) -> float:53 return path.stat().st_size / (1024 * 1024)54 55 56def _print_file_line(label: str, path: Path) -> bool:57 if not path.is_file():58 print(f" {label}: MISSING (expected {path})")59 return False60 try:61 mb = _size_mb(path)62 except OSError as exc:63 print(f" {label}: ERROR stat {path}: {exc}")64 return False65 print(f" {label}: OK")66 print(f" path: {path}")67 print(f" size: {mb:.4f} MB")68 return True69 70 71def main() -> int:72 print("=== Project verification (paths resolved from this script) ===\n")73 print(f"This script: {Path(__file__).resolve()}")74 print(f"Repo root: {REPO_ROOT}")75 print(f"Data dir: {DATA_DIR}")76 print(f"Artifacts: {ARTIFACTS_DIR}\n")77 78 issues: list[str] = []79 80 # 1. Folders81 print("--- Folders ---")82 for label, p in [("backend/data/", DATA_DIR), ("backend/artifacts/", ARTIFACTS_DIR)]:83 if p.is_dir():84 print(f" {label} OK: {p}")85 else:86 print(f" {label} MISSING or not a directory: {p}")87 issues.append(f"folder missing: {p}")88 print()89 90 # 2–3. Expected files91 print("--- Expected files ---")92 all_files_ok = True93 for name, path in EXPECTED_FILES:94 if not _print_file_line(name, path):95 all_files_ok = False96 issues.append(f"missing file: {name}")97 print()98 if not all_files_ok:99 print("--- Backend ready for React frontend integration: NO ---")100 print("Reason: one or more expected files are missing.")101 for i in issues:102 print(f" - {i}")103 return 1104 105 # 4. Dataset columns, row counts, vector alignment (use same artifacts as FastAPI)106 print("--- Dataset & vectors validation ---")107 df_path = ARTIFACTS_DIR / "chatbot_df.pkl"108 v_path = ARTIFACTS_DIR / "vectorizer.pkl"109 qv_path = ARTIFACTS_DIR / "question_vectors.pkl"110 111 try:112 df = joblib.load(df_path)113 if not isinstance(df, pd.DataFrame):114 raise TypeError(f"chatbot_df.pkl must be DataFrame, got {type(df).__name__}")115 Question_vectors = joblib.load(qv_path)116 except Exception as exc:117 print(f"ERROR: failed to load dataframe or vectors: {exc}", file=sys.stderr)118 traceback.print_exc()119 print("\n--- Backend ready for React frontend integration: NO ---")120 return 1121 122 missing_cols = [c for c in REQUIRED_DATASET_COLUMNS if c not in df.columns]123 if missing_cols:124 print(f" FAIL: missing columns: {missing_cols}")125 issues.append(f"missing columns: {missing_cols}")126 else:127 print(f" OK: required columns present ({len(REQUIRED_DATASET_COLUMNS)} fields)")128 129 n_rows = len(df)130 print(f" Dataframe row count: {n_rows:,}")131 132 try:133 n_vec = int(Question_vectors.shape[0])134 except Exception as exc:135 print(f" FAIL: question_vectors shape: {exc}")136 issues.append("invalid question_vectors")137 n_vec = -1138 139 if n_vec >= 0:140 print(f" Question_vectors rows: {n_vec:,}")141 if n_vec == n_rows:142 print(" OK: vector row count matches dataframe row count")143 else:144 print(145 f" FAIL: row mismatch (dataframe {n_rows:,} vs vectors {n_vec:,})"146 )147 issues.append("vector rows != dataframe rows")148 149 if missing_cols or n_vec != n_rows or n_vec < 0:150 print("\n--- Backend ready for React frontend integration: NO ---")151 for i in issues:152 print(f" - {i}")153 return 1154 155 print()156 157 # 5. One test query (same pipeline as retriever / test_artifacts)158 print(f"--- Test query ---")159 print(f" Query: {TEST_QUERY!r}\n")160 try:161 vectorizer = joblib.load(v_path)162 input_question = BeautifulSoup(TEST_QUERY).get_text()163 input_question_vector = vectorizer.transform([input_question])164 similarities = cosine_similarity(input_question_vector, Question_vectors)165 closest = np.argmax(similarities, axis=1)166 row = int(closest[0])167 qid = df.iloc[row, 0]168 answers = df.loc[df["QId"] == qid]169 triple = getAnswerWithHighestScore(answers, df)170 print(f" OK: inference completed (QId={int(qid)}, AId={int(triple[1])})")171 print(f" Answer preview: {str(triple[0])[:200]}{'...' if len(str(triple[0])) > 200 else ''}")172 except Exception as exc:173 print(f" FAIL: {exc}", file=sys.stderr)174 traceback.print_exc()175 print("\n--- Backend ready for React frontend integration: NO ---")176 print("Reason: test query failed.")177 return 1178 179 print()180 print("--- Backend ready for React frontend integration: YES ---")181 print(" All expected artifacts present, dataset validates, vectors align,")182 print(" and a smoke retrieval query succeeded. Start the API with uvicorn")183 print(" (e.g. from backend/: uvicorn app.main:app --host 127.0.0.1 --port 8000).")184 return 0185 186 187if __name__ == "__main__":188 raise SystemExit(main())189 