Team Ai
Apppublic

Decoder2704/python-chatbot-jay

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
check_files.py179 linesDownload Raw Back to debug
1#!/usr/bin/env python32"""3Inspect backend/artifacts and backend/data — read-only; no data modification.4 5Run:6  python backend/debug/check_files.py7"""8 9from __future__ import annotations10 11import sys12from pathlib import Path13 14_BACKEND_DIR = Path(__file__).resolve().parent.parent15if str(_BACKEND_DIR) not in sys.path:16    sys.path.insert(0, str(_BACKEND_DIR))17 18from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT19 20SCAN_TARGETS = [21    ("artifacts", ARTIFACTS_DIR),22    ("data", DATA_DIR),23]24 25 26def _size_mb(path: Path) -> float:27    return path.stat().st_size / (1024 * 1024)28 29 30def _collect_files(root: Path) -> list[Path]:31    if not root.exists() or not root.is_dir():32        return []33    return sorted(p for p in root.rglob("*") if p.is_file())34 35 36def _primary_category(path: Path) -> str:37    """Single bucket per file for the summary report."""38    name = path.name.lower()39    suf = path.suffix.lower()40 41    if suf == ".pkl" and "vectorizer" in name:42        return "vectorizer files (.pkl)"43 44    if suf == ".pkl":45        if (46            "question_vector" in name47            or ("vector" in name and "vectorizer" not in name)48            or "matrix" in name49            or "tfidf" in name50        ):51            return "tfidf / vector files (.pkl)"52        if any(x in name for x in ("chatbot", "final", "dataset")) or name.endswith(53            "_df.pkl"54        ):55            return "dataframe / dataset files (.csv or .pkl)"56        if "data" in name and "vector" not in name:57            return "dataframe / dataset files (.csv or .pkl)"58        return "other (.pkl)"59 60    if suf == ".csv":61        return "dataframe / dataset files (.csv or .pkl)"62 63    return f"other ({suf or 'no extension'})"64 65 66def _likely_vectorizer(paths: list[Path]) -> list[Path]:67    return sorted(p for p in paths if "vectorizer" in p.name.lower())68 69 70def _likely_vectors(paths: list[Path]) -> list[Path]:71    out = []72    for p in paths:73        low = p.name.lower()74        if "vectorizer" in low:75            continue76        if "vector" in low or "matrix" in low:77            out.append(p)78    return sorted(out)79 80 81def _likely_dataset(paths: list[Path]) -> list[Path]:82    return sorted(83        p for p in paths if any(x in p.name.lower() for x in ("data", "df", "chatbot"))84    )85 86 87def _summary_line(paths: list[Path]) -> str:88    if not paths:89        return "<not found>"90    if len(paths) == 1:91        return str(paths[0])92    return f"{paths[0]}  (+ {len(paths) - 1} more)"93 94 95def main() -> int:96    print(f"This script: {Path(__file__).resolve()}")97    print(f"Repo root:   {_REPO_ROOT}")98    print(f"Data dir:    {DATA_DIR}")99    print(f"Artifacts:   {ARTIFACTS_DIR}")100    print("Scanning (read-only): backend/artifacts/, backend/data/\n")101 102    all_files: list[Path] = []103 104    for label, directory in SCAN_TARGETS:105        print(f"=== {label}: {directory} ===")106        if not directory.exists():107            print(108                f"  (folder missing — create it or run the build pipeline; expected: {directory})\n"109            )110            continue111        if not directory.is_dir():112            print(f"  (not a directory — skipped: {directory})\n")113            continue114 115        found = _collect_files(directory)116        if not found:117            print("  (no files)\n")118            continue119 120        for f in found:121            all_files.append(f)122            try:123                mb = _size_mb(f)124            except OSError as exc:125                print(f"  ERROR stat {f}: {exc}", file=sys.stderr)126                continue127            print(f"  File: {f.name}")128            print(f"    Path: {f}")129            print(f"    Size: {mb:.4f} MB")130            print()131 132    # Categorization133    by_cat: dict[str, list[Path]] = {}134    for p in all_files:135        cat = _primary_category(p)136        by_cat.setdefault(cat, []).append(p)137 138    print("=== Categorization ===")139    if not all_files:140        print("  (no files to categorize)\n")141    else:142        for cat in sorted(by_cat.keys()):143            print(f"  {cat}:")144            for p in sorted(by_cat[cat], key=lambda x: str(x)):145                print(f"    - {p}")146            print()147 148    lv = _likely_vectorizer(all_files)149    lvec = _likely_vectors(all_files)150    lds = _likely_dataset(all_files)151 152    print("=== Likely role (by filename) ===")153    print(f"  vectorizer (name contains 'vectorizer'): {len(lv)} file(s)")154    for p in lv:155        print(f"    - {p}")156    print(157        f"  vectors (name contains 'vector' or 'matrix', excluding vectorizer): "158        f"{len(lvec)} file(s)"159    )160    for p in lvec:161        print(f"    - {p}")162    print(163        f"  dataset (name contains 'data', 'df', or 'chatbot'): {len(lds)} file(s)"164    )165    for p in lds:166        print(f"    - {p}")167    print()168 169    print("=== Summary ===")170    print(f"  vectorizer: {_summary_line(lv)}")171    print(f"  vectors:    {_summary_line(lvec)}")172    print(f"  dataset:    {_summary_line(lds)}")173 174    return 0175 176 177if __name__ == "__main__":178    raise SystemExit(main())179