Decoder2704/python-chatbot-jay
0
1"""2Chatbot retrieval aligned with Chatbot-V5.ipynb.3 4Helpers (same names and behavior as the notebook):5 getAnswerWithHighestScore, get_answer, chatbot_response6 7Artifacts map to notebook globals: vectorizer, Question_vectors, df.8The notebook also uses `an` (AnswersV2) for CSV writes in the date-threshold branch;9that frame is not loaded here — only `df` is updated in memory.10"""11 12from __future__ import annotations13 14import logging15from datetime import date, datetime16from pathlib import Path17from typing import Any18 19import joblib20import numpy as np21import pandas as pd22from bs4 import BeautifulSoup23from sklearn.metrics.pairwise import cosine_similarity24 25from ..paths import (26 ARTIFACTS_DIR,27 ARTIFACT_CHATBOT_DF_PATH,28 ARTIFACT_QUESTION_VECTORS_PATH,29 ARTIFACT_VECTORIZER_PATH,30 BACKEND_DIR,31 DATA_DIR,32 RETRIEVAL_ARTIFACT_FILES,33)34 35logger = logging.getLogger(__name__)36 37# Re-export for callers that use retriever.ARTIFACTS_DIR / retriever.DATA_DIR38 39THRESHOLD = 3040 41_vectorizer: Any = None42_question_vectors: Any = None43_df: pd.DataFrame | None = None44 45 46def load_artifacts() -> None:47 """Load vectorizer, Question_vectors, and df from backend/artifacts (startup)."""48 global _vectorizer, _question_vectors, _df49 50 backend_root = BACKEND_DIR.resolve()51 artifacts_root = ARTIFACTS_DIR.resolve()52 data_root = DATA_DIR.resolve()53 54 logger.info(55 "Path anchor: app.paths from %s -> backend root %s (not cwd)",56 Path(__file__).resolve(),57 backend_root,58 )59 logger.info("Artifacts directory (absolute): %s", artifacts_root)60 logger.info("Data directory (absolute): %s", data_root)61 for logical_name, abs_path in RETRIEVAL_ARTIFACT_FILES:62 logger.info(63 "Expected retrieval artifact: %s -> %s",64 logical_name,65 abs_path,66 )67 68 missing = [(name, p) for name, p in RETRIEVAL_ARTIFACT_FILES if not p.is_file()]69 if missing:70 lines = "\n".join(f" - {p} ({name})" for name, p in missing)71 raise RuntimeError(72 "Missing required retrieval artifact file(s). "73 "Paths are resolved from backend/app/paths.py (not the process working directory).\n"74 f"{lines}\n\n"75 f"backend root: {backend_root}\n"76 f"artifacts dir: {artifacts_root}\n"77 "Ensure vectorizer.pkl, question_vectors.pkl, and chatbot_df.pkl are present. "78 "Generate them with: backend/scripts/build_retrieval_artifacts.py "79 "(after build_answers_v2.py and build_final_dataset.py)."80 )81 82 v_path = ARTIFACT_VECTORIZER_PATH83 qv_path = ARTIFACT_QUESTION_VECTORS_PATH84 df_path = ARTIFACT_CHATBOT_DF_PATH85 86 logger.info("Loading joblib: %s", v_path)87 _vectorizer = joblib.load(str(v_path))88 logger.info("Loading joblib: %s", qv_path)89 _question_vectors = joblib.load(str(qv_path))90 logger.info("Loading joblib: %s", df_path)91 _df = joblib.load(str(df_path))92 assert _df is not None93 logger.info(94 "Loaded vectorizer, Question_vectors (shape=%s), chatbot_df (%s rows)",95 getattr(_question_vectors, "shape", "?"),96 f"{len(_df):,}",97 )98 99 100def _require_artifacts() -> tuple[Any, Any, pd.DataFrame]:101 if _vectorizer is None or _question_vectors is None or _df is None:102 raise RuntimeError("Artifacts not loaded; call load_artifacts() at startup")103 return _vectorizer, _question_vectors, _df104 105 106def getAnswerWithHighestScore(answers: pd.DataFrame, df: pd.DataFrame) -> list:107 """108 Notebook getAnswerWithHighestScore (cell 17).109 110 Uses columns: latest_score, alternate, date, QId; mutates `df` when the date111 threshold is exceeded (same intent as notebook; `an` / CSV not available here).112 """113 r = np.argmax(answers["latest_score"] - answers["alternate"])114 date_cell = answers["date"].iloc[0]115 if isinstance(date_cell, str):116 parsed = datetime.strptime(date_cell, "%Y-%m-%d")117 else:118 parsed = datetime.strptime(str(date_cell)[:10], "%Y-%m-%d")119 120 if (121 datetime.strptime(str(date.today()), "%Y-%m-%d") - parsed122 ).days > THRESHOLD:123 qid0 = answers["QId"].iloc[0]124 mask = df["QId"] == qid0125 df.loc[mask, "date"] = date.today()126 df.loc[mask, "alternate"] = df.loc[mask, "latest_score"]127 128 return [129 answers.iloc[r]["Answer"],130 answers.iloc[r]["AId"],131 answers.iloc[r]["alternate"],132 ]133 134 135def get_answer(row: int, df: pd.DataFrame) -> list:136 """137 Notebook get_answer (cell 18).138 139 Returns [answer[0], answer[2]] from getAnswerWithHighestScore (answer text,140 alternate score), matching the notebook return value.141 """142 qid = df.iloc[row, 0]143 answers = df.loc[df["QId"] == qid]144 answer = getAnswerWithHighestScore(answers, df)145 return [answer[0], answer[2]]146 147 148def chatbot_response(msg: str) -> list:149 """150 Notebook chatbot_response (cell 19).151 152 Returns [a[0], a[1]] where `a` is the return of get_answer (answer text, alternate).153 """154 vectorizer, Question_vectors, df = _require_artifacts()155 156 input_question = BeautifulSoup(msg).get_text()157 158 input_question_vector = vectorizer.transform([input_question])159 160 similarities = cosine_similarity(input_question_vector, Question_vectors)161 162 closest = np.argmax(similarities, axis=1)163 a = get_answer(int(closest[0]), df)164 return [a[0], a[1]]165 166 167def chatbot_response_for_api(msg: str) -> tuple[Any, Any, int, int]:168 """169 Same inference path as the notebook (one TF-IDF pass + one getAnswerWithHighestScore).170 171 Returns (answer, alternate, qid, aid) for FastAPI ChatResponse without calling172 get_answer/chatbot_response twice (avoids duplicate score / date updates).173 """174 vectorizer, Question_vectors, df = _require_artifacts()175 176 input_question = BeautifulSoup(msg).get_text()177 input_question_vector = vectorizer.transform([input_question])178 similarities = cosine_similarity(input_question_vector, Question_vectors)179 closest = np.argmax(similarities, axis=1)180 row = int(closest[0])181 182 qid = df.iloc[row, 0]183 answers = df.loc[df["QId"] == qid]184 triple = getAnswerWithHighestScore(answers, df)185 return triple[0], triple[2], int(qid), int(triple[1])186 