Decoder2704/python-chatbot-jay
0
1#!/usr/bin/env python32"""3Inspect and print the structure of data files in a folder.4 5CSV/TSV uses the standard library only (no pandas), so it runs in minimal6Python installs (e.g. Code Runner) and streams large files without loading7them entirely into memory.8 9Usage:10 python view_data_structure.py11 python view_data_structure.py --data-dir "pybot_data"12 python view_data_structure.py --data-dir "/full/path/to/data"13"""14 15from __future__ import annotations16 17import argparse18import csv19import json20from pathlib import Path21from typing import Iterable22 23 24SUPPORTED_EXTENSIONS = {".csv", ".json", ".txt", ".tsv"}25 26 27def format_size(size_in_bytes: int) -> str:28 units = ["B", "KB", "MB", "GB", "TB"]29 size = float(size_in_bytes)30 for unit in units:31 if size < 1024 or unit == units[-1]:32 return f"{size:.2f} {unit}"33 size /= 102434 return f"{size_in_bytes} B"35 36 37def get_files(data_dir: Path) -> Iterable[Path]:38 return sorted(39 [p for p in data_dir.iterdir() if p.is_file() and p.suffix.lower() in SUPPORTED_EXTENSIONS],40 key=lambda p: p.name.lower(),41 )42 43 44def _truncate_cell(value: str, max_len: int = 120) -> str:45 value = value.replace("\n", "\\n").replace("\r", "\\r")46 if len(value) <= max_len:47 return value48 return value[: max_len - 3] + "..."49 50 51def inspect_csv_tsv(path: Path, delimiter: str) -> None:52 """Stream CSV/TSV with stdlib only (no pandas) — works in minimal Python envs."""53 encodings = ("utf-8", "latin-1", "cp1252")54 last_decode_error: UnicodeDecodeError | None = None55 used_encoding: str | None = None56 header: list[str] = []57 num_cols = 058 data_rows = 059 sample_rows: list[list[str]] = []60 61 for encoding in encodings:62 try:63 with path.open("r", encoding=encoding, newline="") as f:64 reader = csv.reader(f, delimiter=delimiter)65 try:66 header = next(reader)67 except StopIteration:68 header = []69 num_cols = 070 data_rows = 071 sample_rows = []72 used_encoding = encoding73 break74 num_cols = len(header)75 sample_rows = []76 data_rows = 077 for row in reader:78 data_rows += 179 if len(sample_rows) < 3:80 sample_rows.append(row)81 used_encoding = encoding82 break83 except UnicodeDecodeError as exc:84 last_decode_error = exc85 continue86 87 if used_encoding is None:88 raise RuntimeError(89 f"Could not decode with encodings {encodings}: {last_decode_error}"90 ) from last_decode_error91 92 if used_encoding != "utf-8":93 print(f" - Encoding: {used_encoding} (fallback)")94 95 print(f" - Shape: {data_rows} data rows x {num_cols} columns")96 print(" - Columns:")97 for col in header:98 print(f" * {col}")99 print(" - Sample rows (cells truncated for display):")100 if not sample_rows and data_rows == 0:101 print(" (no data rows)")102 for i, row in enumerate(sample_rows, start=1):103 shown = [_truncate_cell(c) for c in row]104 print(f" Row {i}: {shown}")105 106 107def inspect_json(path: Path) -> None:108 with path.open("r", encoding="utf-8") as f:109 data = json.load(f)110 111 if isinstance(data, list):112 print(f" - Top-level type: list ({len(data)} items)")113 if data and isinstance(data[0], dict):114 keys = sorted(data[0].keys())115 print(f" - Item type: object with keys: {', '.join(keys)}")116 elif data:117 print(f" - Item type: {type(data[0]).__name__}")118 elif isinstance(data, dict):119 keys = sorted(data.keys())120 print(f" - Top-level type: object with {len(keys)} keys")121 print(f" - Keys: {', '.join(keys[:20])}{' ...' if len(keys) > 20 else ''}")122 else:123 print(f" - Top-level type: {type(data).__name__}")124 125 126def inspect_txt(path: Path) -> None:127 with path.open("r", encoding="utf-8", errors="replace") as f:128 lines = f.readlines()129 print(f" - Lines: {len(lines)}")130 preview = "".join(lines[:3]).strip()131 if preview:132 print(" - Preview:")133 for line in preview.splitlines():134 print(f" {line}")135 136 137def inspect_file(path: Path) -> None:138 print(f"\nFile: {path.name}")139 print(f" - Type: {path.suffix.lower()[1:] or 'unknown'}")140 print(f" - Size: {format_size(path.stat().st_size)}")141 142 suffix = path.suffix.lower()143 try:144 if suffix == ".csv":145 inspect_csv_tsv(path, delimiter=",")146 elif suffix == ".tsv":147 inspect_csv_tsv(path, delimiter="\t")148 elif suffix == ".json":149 inspect_json(path)150 elif suffix == ".txt":151 inspect_txt(path)152 else:153 print(" - No inspector implemented for this format.")154 except Exception as exc: # noqa: BLE001 - user-facing utility script155 print(f" - Could not inspect file details: {exc}")156 157 158def parse_args() -> argparse.Namespace:159 parser = argparse.ArgumentParser(description="View data structure in a dataset folder.")160 parser.add_argument(161 "--data-dir",162 default="pybot_data",163 help="Path to data folder (default: pybot_data)",164 )165 return parser.parse_args()166 167 168def main() -> None:169 args = parse_args()170 data_dir = Path(args.data_dir).expanduser().resolve()171 172 print("=" * 70)173 print("PyBot Data Structure Viewer")174 print("=" * 70)175 print(f"Data directory: {data_dir}")176 177 if not data_dir.exists():178 print("\nThe data directory does not exist.")179 print("Create it or pass another path with --data-dir.")180 return181 182 if not data_dir.is_dir():183 print("\nThe provided path is not a directory.")184 return185 186 files = list(get_files(data_dir))187 if not files:188 print("\nNo supported data files found.")189 print("Supported extensions: .csv, .tsv, .json, .txt")190 return191 192 print(f"\nFound {len(files)} supported file(s).")193 for file_path in files:194 inspect_file(file_path)195 196 print("\nDone.")197 198 199if __name__ == "__main__":200 main()201 