Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Verify TF-IDF artifacts (read-only on disk; may update in-memory df like production).4 5Uses the same retrieval steps as Chatbot-V5.ipynb / retriever: transform, cosine6similarity, argmax, then getAnswerWithHighestScore for the matched QId group.7 8Run:9 python backend/debug/test_artifacts.py10"""11 12from __future__ import annotations13 14import sys15import traceback16from pathlib import Path17 18import joblib19import numpy as np20import pandas as pd21from bs4 import BeautifulSoup22from sklearn.metrics.pairwise import cosine_similarity23 24_BACKEND_DIR = Path(__file__).resolve().parent.parent25if str(_BACKEND_DIR) not in sys.path:26 sys.path.insert(0, str(_BACKEND_DIR))27 28from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT29 30from app.services.retriever import getAnswerWithHighestScore # noqa: E40231 32TEST_QUERY = "What is list comprehension in Python?"33ANSWER_TRUNCATE = 20034 35 36def _find_vectorizer_path() -> Path:37 p = ARTIFACTS_DIR / "vectorizer.pkl"38 if p.is_file():39 return p40 raise FileNotFoundError(41 f"vectorizer.pkl not found. Expected: {p}\n"42 f" (artifacts dir exists: {ARTIFACTS_DIR.is_dir()}, path: {ARTIFACTS_DIR})"43 )44 45 46def _find_question_vectors_path() -> Path:47 preferred = ARTIFACTS_DIR / "question_vectors.pkl"48 if preferred.is_file():49 return preferred50 if ARTIFACTS_DIR.is_dir():51 for p in sorted(ARTIFACTS_DIR.glob("*.pkl")):52 low = p.name.lower()53 if "vectorizer" in low:54 continue55 if "vector" in low or "matrix" in low or "tfidf" in low:56 return p57 raise FileNotFoundError(58 f"No question_vectors (or similar) .pkl found under {ARTIFACTS_DIR} "59 f"(exists: {ARTIFACTS_DIR.is_dir()})\n"60 " Run: backend/scripts/build_retrieval_artifacts.py"61 )62 63 64def _find_dataset_path() -> Path:65 p = ARTIFACTS_DIR / "chatbot_df.pkl"66 if p.is_file():67 return p68 csv_path = DATA_DIR / "final_chatbot_data.csv"69 if csv_path.is_file():70 return csv_path71 if DATA_DIR.is_dir():72 csvs = sorted(DATA_DIR.glob("*.csv"))73 if csvs:74 return csvs[0]75 raise FileNotFoundError(76 f"No chatbot_df.pkl or dataset .csv found.\n"77 f" artifacts: {ARTIFACTS_DIR} (exists: {ARTIFACTS_DIR.is_dir()})\n"78 f" data: {DATA_DIR} (exists: {DATA_DIR.is_dir()})\n"79 " Run the build pipeline under backend/scripts/ first."80 )81 82 83def _load_dataset(path: Path) -> pd.DataFrame:84 if path.suffix.lower() == ".pkl":85 obj = joblib.load(path)86 if not isinstance(obj, pd.DataFrame):87 raise TypeError(f"Expected DataFrame from {path}, got {type(obj).__name__}")88 return obj89 if path.suffix.lower() == ".csv":90 return pd.read_csv(path)91 raise ValueError(f"Unsupported dataset file: {path}")92 93 94def _truncate(text: object, width: int = ANSWER_TRUNCATE) -> str:95 s = str(text)96 if len(s) <= width:97 return s98 return s[: width - 3] + "..."99 100 101def main() -> int:102 print(f"This script: {Path(__file__).resolve()}")103 print(f"Repo root: {_REPO_ROOT}")104 print(f"Data dir: {DATA_DIR}")105 print(f"Artifacts: {ARTIFACTS_DIR}")106 print("TF-IDF artifact verification (read-only files)\n")107 108 try:109 v_path = _find_vectorizer_path()110 qv_path = _find_question_vectors_path()111 ds_path = _find_dataset_path()112 except FileNotFoundError as exc:113 print(f"ERROR: {exc}", file=sys.stderr)114 return 1115 116 print("Resolved paths:")117 print(f" vectorizer: {v_path}")118 print(f" question vecs: {qv_path}")119 print(f" dataset: {ds_path}\n")120 121 try:122 vectorizer = joblib.load(v_path)123 Question_vectors = joblib.load(qv_path)124 df = _load_dataset(ds_path)125 except Exception as exc:126 print(f"ERROR: Failed to load artifacts: {exc}", file=sys.stderr)127 traceback.print_exc()128 return 1129 130 n_rows = len(df)131 if n_rows == 0:132 print("ERROR: Dataset has length 0", file=sys.stderr)133 return 1134 135 try:136 n_vec = Question_vectors.shape[0]137 except Exception as exc:138 print(f"ERROR: question_vectors has no shape: {exc}", file=sys.stderr)139 return 1140 141 if n_vec != n_rows:142 print(143 f"ERROR: Row mismatch — dataset has {n_rows:,} rows but "144 f"question_vectors has {n_vec:,} rows (first dimension).",145 file=sys.stderr,146 )147 return 1148 149 print(f"Validation: dataset rows = {n_rows:,}, question_vectors.shape[0] = {n_vec:,} (OK)")150 151 try:152 _ = vectorizer.transform(["sanity check transform"])153 except Exception as exc:154 print(f"ERROR: vectorizer.transform failed: {exc}", file=sys.stderr)155 traceback.print_exc()156 return 1157 158 print("Validation: vectorizer.transform() OK\n")159 160 print(f"Test query: {TEST_QUERY!r}\n")161 162 try:163 input_question = BeautifulSoup(TEST_QUERY).get_text()164 input_question_vector = vectorizer.transform([input_question])165 similarities = cosine_similarity(input_question_vector, Question_vectors)166 closest = np.argmax(similarities, axis=1)167 row = int(closest[0])168 169 qid = df.iloc[row, 0]170 answers = df.loc[df["QId"] == qid]171 triple = getAnswerWithHighestScore(answers, df)172 answer_text = triple[0]173 aid = int(triple[1])174 except Exception as exc:175 print(f"ERROR: Inference failed: {exc}", file=sys.stderr)176 traceback.print_exc()177 return 1178 179 print("Top match (same pipeline as notebook / retriever):")180 print(f" Matched QId: {int(qid)}")181 print(f" Matched AId: {aid}")182 print(f" Answer (truncated): {_truncate(answer_text)}")183 184 return 0185 186 187if __name__ == "__main__":188 raise SystemExit(main())189 