Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
test_pipeline.py211 linesDownload Raw Back to debug
1#!/usr/bin/env python32"""3Simulate the full chatbot pipeline: load artifacts, preprocess, TF-IDF, similarity,4answer selection (same flow as retriever / notebook).5 6Does not modify application code or retriever modules.7 8Run:9  python backend/debug/test_pipeline.py10"""11 12from __future__ import annotations13 14import sys15import traceback16from pathlib import Path17 18import joblib19import numpy as np20import pandas as pd21from bs4 import BeautifulSoup22from sklearn.metrics.pairwise import cosine_similarity23 24_BACKEND_DIR = Path(__file__).resolve().parent.parent25if str(_BACKEND_DIR) not in sys.path:26    sys.path.insert(0, str(_BACKEND_DIR))27 28from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT29 30from app.services.retriever import getAnswerWithHighestScore31 32TEXT_TRUNCATE = 15033 34# Topics: list comprehension, file handling, exceptions, lambda functions, dictionaries35TEST_QUERIES = [36    "How do I use list comprehension in Python?",37    "How can I read and write files in Python?",38    "How do I handle exceptions with try and except in Python?",39    "What are lambda functions in Python and when should I use them?",40    "How do I iterate over keys and values in a dictionary in Python?",41]42 43 44def _find_vectorizer_path() -> Path:45    p = ARTIFACTS_DIR / "vectorizer.pkl"46    if p.is_file():47        return p48    raise FileNotFoundError(49        f"vectorizer.pkl not found. Expected: {p}\n"50        f"  (artifacts dir exists: {ARTIFACTS_DIR.is_dir()}, path: {ARTIFACTS_DIR})"51    )52 53 54def _find_question_vectors_path() -> Path:55    preferred = ARTIFACTS_DIR / "question_vectors.pkl"56    if preferred.is_file():57        return preferred58    if ARTIFACTS_DIR.is_dir():59        for p in sorted(ARTIFACTS_DIR.glob("*.pkl")):60            low = p.name.lower()61            if "vectorizer" in low:62                continue63            if "vector" in low or "matrix" in low or "tfidf" in low:64                return p65    raise FileNotFoundError(66        f"No question_vectors (or similar) .pkl found under {ARTIFACTS_DIR} "67        f"(exists: {ARTIFACTS_DIR.is_dir()})\n"68        "  Run: backend/scripts/build_retrieval_artifacts.py"69    )70 71 72def _find_dataset_path() -> Path:73    p = ARTIFACTS_DIR / "chatbot_df.pkl"74    if p.is_file():75        return p76    csv_path = DATA_DIR / "final_chatbot_data.csv"77    if csv_path.is_file():78        return csv_path79    if DATA_DIR.is_dir():80        csvs = sorted(DATA_DIR.glob("*.csv"))81        if csvs:82            return csvs[0]83    raise FileNotFoundError(84        f"No chatbot_df.pkl or dataset .csv found.\n"85        f"  artifacts: {ARTIFACTS_DIR} (exists: {ARTIFACTS_DIR.is_dir()})\n"86        f"  data:      {DATA_DIR} (exists: {DATA_DIR.is_dir()})\n"87        "  Run the build pipeline under backend/scripts/ first."88    )89 90 91def _load_dataset(path: Path) -> pd.DataFrame:92    if path.suffix.lower() == ".pkl":93        obj = joblib.load(path)94        if not isinstance(obj, pd.DataFrame):95            raise TypeError(f"Expected DataFrame from {path}, got {type(obj).__name__}")96        return obj97    if path.suffix.lower() == ".csv":98        return pd.read_csv(path)99    raise ValueError(f"Unsupported dataset file: {path}")100 101 102def _truncate(text: object, width: int = TEXT_TRUNCATE) -> str:103    s = str(text)104    if len(s) <= width:105        return s106    return s[: width - 3] + "..."107 108 109def _run_one_query(110    raw_input: str,111    vectorizer,112    Question_vectors,113    df: pd.DataFrame,114) -> tuple[str, str, str]:115    """116    Preprocess → vectorize → cosine similarity → top row → getAnswerWithHighestScore.117    Returns (matched_question_text, answer_text, error_message_or_empty).118    """119    try:120        input_question = BeautifulSoup(raw_input).get_text()121        input_question_vector = vectorizer.transform([input_question])122        similarities = cosine_similarity(input_question_vector, Question_vectors)123        closest = np.argmax(similarities, axis=1)124        row = int(closest[0])125 126        matched_question = df.iloc[row]["Question"]127        qid = df.iloc[row, 0]128        answers = df.loc[df["QId"] == qid]129        triple = getAnswerWithHighestScore(answers, df)130        answer_text = triple[0]131        return str(matched_question), str(answer_text), ""132    except Exception as exc:133        traceback.print_exc()134        return "", "", f"{type(exc).__name__}: {exc}"135 136 137def main() -> int:138    print(f"This script: {Path(__file__).resolve()}")139    print(f"Repo root:   {_REPO_ROOT}")140    print(f"Data dir:    {DATA_DIR}")141    print(f"Artifacts:   {ARTIFACTS_DIR}")142    print("Full pipeline simulation (dataset + vectorizer + question_vectors)\n")143 144    try:145        v_path = _find_vectorizer_path()146        qv_path = _find_question_vectors_path()147        ds_path = _find_dataset_path()148    except FileNotFoundError as exc:149        print(f"ERROR: {exc}", file=sys.stderr)150        return 1151 152    print("Resolved paths:")153    print(f"  vectorizer:     {v_path}")154    print(f"  question vecs:  {qv_path}")155    print(f"  dataset:        {ds_path}\n")156 157    try:158        vectorizer = joblib.load(v_path)159        Question_vectors = joblib.load(qv_path)160        df = _load_dataset(ds_path)161    except Exception as exc:162        print(f"ERROR: Failed to load artifacts: {exc}", file=sys.stderr)163        traceback.print_exc()164        return 1165 166    if len(df) == 0:167        print("ERROR: Dataset is empty.", file=sys.stderr)168        return 1169 170    try:171        nv = Question_vectors.shape[0]172    except Exception as exc:173        print(f"ERROR: question_vectors invalid: {exc}", file=sys.stderr)174        return 1175 176    if nv != len(df):177        print(178            f"ERROR: question_vectors rows ({nv}) != dataset rows ({len(df)}).",179            file=sys.stderr,180        )181        return 1182 183    if "Question" not in df.columns:184        print("ERROR: Dataset missing 'Question' column.", file=sys.stderr)185        return 1186 187    failures = 0188    for i, query in enumerate(TEST_QUERIES, start=1):189        print(f"========== Query {i}/5 ==========")190        print(f"Input:\n  {query}\n")191 192        mq, ans, err = _run_one_query(query, vectorizer, Question_vectors, df)193        if err:194            print(f"ERROR: {err}\n", file=sys.stderr)195            failures += 1196            continue197 198        print(f"Matched question (truncated):\n  {_truncate(mq)}\n")199        print(f"Answer (truncated):\n  {_truncate(ans)}\n")200 201    if failures:202        print(f"Completed with {failures} failed query(s).", file=sys.stderr)203        return 1204 205    print("Pipeline run finished (all queries OK).")206    return 0207 208 209if __name__ == "__main__":210    raise SystemExit(main())211