Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
test_artifacts.py189 linesDownload Raw Back to debug
1#!/usr/bin/env python32"""3Verify TF-IDF artifacts (read-only on disk; may update in-memory df like production).4 5Uses the same retrieval steps as Chatbot-V5.ipynb / retriever: transform, cosine6similarity, argmax, then getAnswerWithHighestScore for the matched QId group.7 8Run:9  python backend/debug/test_artifacts.py10"""11 12from __future__ import annotations13 14import sys15import traceback16from pathlib import Path17 18import joblib19import numpy as np20import pandas as pd21from bs4 import BeautifulSoup22from sklearn.metrics.pairwise import cosine_similarity23 24_BACKEND_DIR = Path(__file__).resolve().parent.parent25if str(_BACKEND_DIR) not in sys.path:26    sys.path.insert(0, str(_BACKEND_DIR))27 28from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT29 30from app.services.retriever import getAnswerWithHighestScore  # noqa: E40231 32TEST_QUERY = "What is list comprehension in Python?"33ANSWER_TRUNCATE = 20034 35 36def _find_vectorizer_path() -> Path:37    p = ARTIFACTS_DIR / "vectorizer.pkl"38    if p.is_file():39        return p40    raise FileNotFoundError(41        f"vectorizer.pkl not found. Expected: {p}\n"42        f"  (artifacts dir exists: {ARTIFACTS_DIR.is_dir()}, path: {ARTIFACTS_DIR})"43    )44 45 46def _find_question_vectors_path() -> Path:47    preferred = ARTIFACTS_DIR / "question_vectors.pkl"48    if preferred.is_file():49        return preferred50    if ARTIFACTS_DIR.is_dir():51        for p in sorted(ARTIFACTS_DIR.glob("*.pkl")):52            low = p.name.lower()53            if "vectorizer" in low:54                continue55            if "vector" in low or "matrix" in low or "tfidf" in low:56                return p57    raise FileNotFoundError(58        f"No question_vectors (or similar) .pkl found under {ARTIFACTS_DIR} "59        f"(exists: {ARTIFACTS_DIR.is_dir()})\n"60        "  Run: backend/scripts/build_retrieval_artifacts.py"61    )62 63 64def _find_dataset_path() -> Path:65    p = ARTIFACTS_DIR / "chatbot_df.pkl"66    if p.is_file():67        return p68    csv_path = DATA_DIR / "final_chatbot_data.csv"69    if csv_path.is_file():70        return csv_path71    if DATA_DIR.is_dir():72        csvs = sorted(DATA_DIR.glob("*.csv"))73        if csvs:74            return csvs[0]75    raise FileNotFoundError(76        f"No chatbot_df.pkl or dataset .csv found.\n"77        f"  artifacts: {ARTIFACTS_DIR} (exists: {ARTIFACTS_DIR.is_dir()})\n"78        f"  data:      {DATA_DIR} (exists: {DATA_DIR.is_dir()})\n"79        "  Run the build pipeline under backend/scripts/ first."80    )81 82 83def _load_dataset(path: Path) -> pd.DataFrame:84    if path.suffix.lower() == ".pkl":85        obj = joblib.load(path)86        if not isinstance(obj, pd.DataFrame):87            raise TypeError(f"Expected DataFrame from {path}, got {type(obj).__name__}")88        return obj89    if path.suffix.lower() == ".csv":90        return pd.read_csv(path)91    raise ValueError(f"Unsupported dataset file: {path}")92 93 94def _truncate(text: object, width: int = ANSWER_TRUNCATE) -> str:95    s = str(text)96    if len(s) <= width:97        return s98    return s[: width - 3] + "..."99 100 101def main() -> int:102    print(f"This script: {Path(__file__).resolve()}")103    print(f"Repo root:   {_REPO_ROOT}")104    print(f"Data dir:    {DATA_DIR}")105    print(f"Artifacts:   {ARTIFACTS_DIR}")106    print("TF-IDF artifact verification (read-only files)\n")107 108    try:109        v_path = _find_vectorizer_path()110        qv_path = _find_question_vectors_path()111        ds_path = _find_dataset_path()112    except FileNotFoundError as exc:113        print(f"ERROR: {exc}", file=sys.stderr)114        return 1115 116    print("Resolved paths:")117    print(f"  vectorizer:     {v_path}")118    print(f"  question vecs:  {qv_path}")119    print(f"  dataset:        {ds_path}\n")120 121    try:122        vectorizer = joblib.load(v_path)123        Question_vectors = joblib.load(qv_path)124        df = _load_dataset(ds_path)125    except Exception as exc:126        print(f"ERROR: Failed to load artifacts: {exc}", file=sys.stderr)127        traceback.print_exc()128        return 1129 130    n_rows = len(df)131    if n_rows == 0:132        print("ERROR: Dataset has length 0", file=sys.stderr)133        return 1134 135    try:136        n_vec = Question_vectors.shape[0]137    except Exception as exc:138        print(f"ERROR: question_vectors has no shape: {exc}", file=sys.stderr)139        return 1140 141    if n_vec != n_rows:142        print(143            f"ERROR: Row mismatch — dataset has {n_rows:,} rows but "144            f"question_vectors has {n_vec:,} rows (first dimension).",145            file=sys.stderr,146        )147        return 1148 149    print(f"Validation: dataset rows = {n_rows:,}, question_vectors.shape[0] = {n_vec:,} (OK)")150 151    try:152        _ = vectorizer.transform(["sanity check transform"])153    except Exception as exc:154        print(f"ERROR: vectorizer.transform failed: {exc}", file=sys.stderr)155        traceback.print_exc()156        return 1157 158    print("Validation: vectorizer.transform() OK\n")159 160    print(f"Test query: {TEST_QUERY!r}\n")161 162    try:163        input_question = BeautifulSoup(TEST_QUERY).get_text()164        input_question_vector = vectorizer.transform([input_question])165        similarities = cosine_similarity(input_question_vector, Question_vectors)166        closest = np.argmax(similarities, axis=1)167        row = int(closest[0])168 169        qid = df.iloc[row, 0]170        answers = df.loc[df["QId"] == qid]171        triple = getAnswerWithHighestScore(answers, df)172        answer_text = triple[0]173        aid = int(triple[1])174    except Exception as exc:175        print(f"ERROR: Inference failed: {exc}", file=sys.stderr)176        traceback.print_exc()177        return 1178 179    print("Top match (same pipeline as notebook / retriever):")180    print(f"  Matched QId: {int(qid)}")181    print(f"  Matched AId: {aid}")182    print(f"  Answer (truncated): {_truncate(answer_text)}")183 184    return 0185 186 187if __name__ == "__main__":188    raise SystemExit(main())189