Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
retriever.py186 linesDownload Raw Back to services
1"""2Chatbot retrieval aligned with Chatbot-V5.ipynb.3 4Helpers (same names and behavior as the notebook):5  getAnswerWithHighestScore, get_answer, chatbot_response6 7Artifacts map to notebook globals: vectorizer, Question_vectors, df.8The notebook also uses `an` (AnswersV2) for CSV writes in the date-threshold branch;9that frame is not loaded here — only `df` is updated in memory.10"""11 12from __future__ import annotations13 14import logging15from datetime import date, datetime16from pathlib import Path17from typing import Any18 19import joblib20import numpy as np21import pandas as pd22from bs4 import BeautifulSoup23from sklearn.metrics.pairwise import cosine_similarity24 25from ..paths import (26    ARTIFACTS_DIR,27    ARTIFACT_CHATBOT_DF_PATH,28    ARTIFACT_QUESTION_VECTORS_PATH,29    ARTIFACT_VECTORIZER_PATH,30    BACKEND_DIR,31    DATA_DIR,32    RETRIEVAL_ARTIFACT_FILES,33)34 35logger = logging.getLogger(__name__)36 37# Re-export for callers that use retriever.ARTIFACTS_DIR / retriever.DATA_DIR38 39THRESHOLD = 3040 41_vectorizer: Any = None42_question_vectors: Any = None43_df: pd.DataFrame | None = None44 45 46def load_artifacts() -> None:47    """Load vectorizer, Question_vectors, and df from backend/artifacts (startup)."""48    global _vectorizer, _question_vectors, _df49 50    backend_root = BACKEND_DIR.resolve()51    artifacts_root = ARTIFACTS_DIR.resolve()52    data_root = DATA_DIR.resolve()53 54    logger.info(55        "Path anchor: app.paths from %s -> backend root %s (not cwd)",56        Path(__file__).resolve(),57        backend_root,58    )59    logger.info("Artifacts directory (absolute): %s", artifacts_root)60    logger.info("Data directory (absolute):     %s", data_root)61    for logical_name, abs_path in RETRIEVAL_ARTIFACT_FILES:62        logger.info(63            "Expected retrieval artifact: %s -> %s",64            logical_name,65            abs_path,66        )67 68    missing = [(name, p) for name, p in RETRIEVAL_ARTIFACT_FILES if not p.is_file()]69    if missing:70        lines = "\n".join(f"  - {p}  ({name})" for name, p in missing)71        raise RuntimeError(72            "Missing required retrieval artifact file(s). "73            "Paths are resolved from backend/app/paths.py (not the process working directory).\n"74            f"{lines}\n\n"75            f"backend root:    {backend_root}\n"76            f"artifacts dir:   {artifacts_root}\n"77            "Ensure vectorizer.pkl, question_vectors.pkl, and chatbot_df.pkl are present. "78            "Generate them with: backend/scripts/build_retrieval_artifacts.py "79            "(after build_answers_v2.py and build_final_dataset.py)."80        )81 82    v_path = ARTIFACT_VECTORIZER_PATH83    qv_path = ARTIFACT_QUESTION_VECTORS_PATH84    df_path = ARTIFACT_CHATBOT_DF_PATH85 86    logger.info("Loading joblib: %s", v_path)87    _vectorizer = joblib.load(str(v_path))88    logger.info("Loading joblib: %s", qv_path)89    _question_vectors = joblib.load(str(qv_path))90    logger.info("Loading joblib: %s", df_path)91    _df = joblib.load(str(df_path))92    assert _df is not None93    logger.info(94        "Loaded vectorizer, Question_vectors (shape=%s), chatbot_df (%s rows)",95        getattr(_question_vectors, "shape", "?"),96        f"{len(_df):,}",97    )98 99 100def _require_artifacts() -> tuple[Any, Any, pd.DataFrame]:101    if _vectorizer is None or _question_vectors is None or _df is None:102        raise RuntimeError("Artifacts not loaded; call load_artifacts() at startup")103    return _vectorizer, _question_vectors, _df104 105 106def getAnswerWithHighestScore(answers: pd.DataFrame, df: pd.DataFrame) -> list:107    """108    Notebook getAnswerWithHighestScore (cell 17).109 110    Uses columns: latest_score, alternate, date, QId; mutates `df` when the date111    threshold is exceeded (same intent as notebook; `an` / CSV not available here).112    """113    r = np.argmax(answers["latest_score"] - answers["alternate"])114    date_cell = answers["date"].iloc[0]115    if isinstance(date_cell, str):116        parsed = datetime.strptime(date_cell, "%Y-%m-%d")117    else:118        parsed = datetime.strptime(str(date_cell)[:10], "%Y-%m-%d")119 120    if (121        datetime.strptime(str(date.today()), "%Y-%m-%d") - parsed122    ).days > THRESHOLD:123        qid0 = answers["QId"].iloc[0]124        mask = df["QId"] == qid0125        df.loc[mask, "date"] = date.today()126        df.loc[mask, "alternate"] = df.loc[mask, "latest_score"]127 128    return [129        answers.iloc[r]["Answer"],130        answers.iloc[r]["AId"],131        answers.iloc[r]["alternate"],132    ]133 134 135def get_answer(row: int, df: pd.DataFrame) -> list:136    """137    Notebook get_answer (cell 18).138 139    Returns [answer[0], answer[2]] from getAnswerWithHighestScore (answer text,140    alternate score), matching the notebook return value.141    """142    qid = df.iloc[row, 0]143    answers = df.loc[df["QId"] == qid]144    answer = getAnswerWithHighestScore(answers, df)145    return [answer[0], answer[2]]146 147 148def chatbot_response(msg: str) -> list:149    """150    Notebook chatbot_response (cell 19).151 152    Returns [a[0], a[1]] where `a` is the return of get_answer (answer text, alternate).153    """154    vectorizer, Question_vectors, df = _require_artifacts()155 156    input_question = BeautifulSoup(msg).get_text()157 158    input_question_vector = vectorizer.transform([input_question])159 160    similarities = cosine_similarity(input_question_vector, Question_vectors)161 162    closest = np.argmax(similarities, axis=1)163    a = get_answer(int(closest[0]), df)164    return [a[0], a[1]]165 166 167def chatbot_response_for_api(msg: str) -> tuple[Any, Any, int, int]:168    """169    Same inference path as the notebook (one TF-IDF pass + one getAnswerWithHighestScore).170 171    Returns (answer, alternate, qid, aid) for FastAPI ChatResponse without calling172    get_answer/chatbot_response twice (avoids duplicate score / date updates).173    """174    vectorizer, Question_vectors, df = _require_artifacts()175 176    input_question = BeautifulSoup(msg).get_text()177    input_question_vector = vectorizer.transform([input_question])178    similarities = cosine_similarity(input_question_vector, Question_vectors)179    closest = np.argmax(similarities, axis=1)180    row = int(closest[0])181 182    qid = df.iloc[row, 0]183    answers = df.loc[df["QId"] == qid]184    triple = getAnswerWithHighestScore(answers, df)185    return triple[0], triple[2], int(qid), int(triple[1])186