Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Inspect backend/artifacts and backend/data — read-only; no data modification.4 5Run:6 python backend/debug/check_files.py7"""8 9from __future__ import annotations10 11import sys12from pathlib import Path13 14_BACKEND_DIR = Path(__file__).resolve().parent.parent15if str(_BACKEND_DIR) not in sys.path:16 sys.path.insert(0, str(_BACKEND_DIR))17 18from app.paths import ARTIFACTS_DIR, DATA_DIR, REPO_ROOT19 20SCAN_TARGETS = [21 ("artifacts", ARTIFACTS_DIR),22 ("data", DATA_DIR),23]24 25 26def _size_mb(path: Path) -> float:27 return path.stat().st_size / (1024 * 1024)28 29 30def _collect_files(root: Path) -> list[Path]:31 if not root.exists() or not root.is_dir():32 return []33 return sorted(p for p in root.rglob("*") if p.is_file())34 35 36def _primary_category(path: Path) -> str:37 """Single bucket per file for the summary report."""38 name = path.name.lower()39 suf = path.suffix.lower()40 41 if suf == ".pkl" and "vectorizer" in name:42 return "vectorizer files (.pkl)"43 44 if suf == ".pkl":45 if (46 "question_vector" in name47 or ("vector" in name and "vectorizer" not in name)48 or "matrix" in name49 or "tfidf" in name50 ):51 return "tfidf / vector files (.pkl)"52 if any(x in name for x in ("chatbot", "final", "dataset")) or name.endswith(53 "_df.pkl"54 ):55 return "dataframe / dataset files (.csv or .pkl)"56 if "data" in name and "vector" not in name:57 return "dataframe / dataset files (.csv or .pkl)"58 return "other (.pkl)"59 60 if suf == ".csv":61 return "dataframe / dataset files (.csv or .pkl)"62 63 return f"other ({suf or 'no extension'})"64 65 66def _likely_vectorizer(paths: list[Path]) -> list[Path]:67 return sorted(p for p in paths if "vectorizer" in p.name.lower())68 69 70def _likely_vectors(paths: list[Path]) -> list[Path]:71 out = []72 for p in paths:73 low = p.name.lower()74 if "vectorizer" in low:75 continue76 if "vector" in low or "matrix" in low:77 out.append(p)78 return sorted(out)79 80 81def _likely_dataset(paths: list[Path]) -> list[Path]:82 return sorted(83 p for p in paths if any(x in p.name.lower() for x in ("data", "df", "chatbot"))84 )85 86 87def _summary_line(paths: list[Path]) -> str:88 if not paths:89 return "<not found>"90 if len(paths) == 1:91 return str(paths[0])92 return f"{paths[0]} (+ {len(paths) - 1} more)"93 94 95def main() -> int:96 print(f"This script: {Path(__file__).resolve()}")97 print(f"Repo root: {_REPO_ROOT}")98 print(f"Data dir: {DATA_DIR}")99 print(f"Artifacts: {ARTIFACTS_DIR}")100 print("Scanning (read-only): backend/artifacts/, backend/data/\n")101 102 all_files: list[Path] = []103 104 for label, directory in SCAN_TARGETS:105 print(f"=== {label}: {directory} ===")106 if not directory.exists():107 print(108 f" (folder missing — create it or run the build pipeline; expected: {directory})\n"109 )110 continue111 if not directory.is_dir():112 print(f" (not a directory — skipped: {directory})\n")113 continue114 115 found = _collect_files(directory)116 if not found:117 print(" (no files)\n")118 continue119 120 for f in found:121 all_files.append(f)122 try:123 mb = _size_mb(f)124 except OSError as exc:125 print(f" ERROR stat {f}: {exc}", file=sys.stderr)126 continue127 print(f" File: {f.name}")128 print(f" Path: {f}")129 print(f" Size: {mb:.4f} MB")130 print()131 132 # Categorization133 by_cat: dict[str, list[Path]] = {}134 for p in all_files:135 cat = _primary_category(p)136 by_cat.setdefault(cat, []).append(p)137 138 print("=== Categorization ===")139 if not all_files:140 print(" (no files to categorize)\n")141 else:142 for cat in sorted(by_cat.keys()):143 print(f" {cat}:")144 for p in sorted(by_cat[cat], key=lambda x: str(x)):145 print(f" - {p}")146 print()147 148 lv = _likely_vectorizer(all_files)149 lvec = _likely_vectors(all_files)150 lds = _likely_dataset(all_files)151 152 print("=== Likely role (by filename) ===")153 print(f" vectorizer (name contains 'vectorizer'): {len(lv)} file(s)")154 for p in lv:155 print(f" - {p}")156 print(157 f" vectors (name contains 'vector' or 'matrix', excluding vectorizer): "158 f"{len(lvec)} file(s)"159 )160 for p in lvec:161 print(f" - {p}")162 print(163 f" dataset (name contains 'data', 'df', or 'chatbot'): {len(lds)} file(s)"164 )165 for p in lds:166 print(f" - {p}")167 print()168 169 print("=== Summary ===")170 print(f" vectorizer: {_summary_line(lv)}")171 print(f" vectors: {_summary_line(lvec)}")172 print(f" dataset: {_summary_line(lds)}")173 174 return 0175 176 177if __name__ == "__main__":178 raise SystemExit(main())179 