kyorlin2001/code-analysis-tool
0
1from __future__ import annotations2 3import re4from dataclasses import dataclass5from pathlib import Path6from typing import Any7 8 9IMPORT_PATTERNS = {10 ".py": re.compile(r"^\s*(?:from\s+\S+\s+import\s+\S+|import\s+\S+)", re.MULTILINE),11 ".js": re.compile(r"^\s*import\s+.*|^\s*const\s+\w+\s*=\s*require\(", re.MULTILINE),12 ".ts": re.compile(r"^\s*import\s+.*|^\s*const\s+\w+\s*=\s*require\(", re.MULTILINE),13 ".jsx": re.compile(r"^\s*import\s+.*|^\s*const\s+\w+\s*=\s*require\(", re.MULTILINE),14 ".tsx": re.compile(r"^\s*import\s+.*|^\s*const\s+\w+\s*=\s*require\(", re.MULTILINE),15}16 17 18@dataclass19class ParsedCodeInfo:20 path: str21 language: str22 line_count: int23 imports: list[str]24 todos: list[str]25 classes: list[str]26 functions: list[str]27 28 29def read_text(path: Path) -> str:30 try:31 return path.read_text(encoding="utf-8", errors="ignore")32 except Exception:33 return ""34 35 36def detect_language(path: str) -> str:37 suffix = Path(path).suffix.lower()38 39 mapping = {40 ".py": "Python",41 ".js": "JavaScript",42 ".jsx": "JavaScript React",43 ".ts": "TypeScript",44 ".tsx": "TypeScript React",45 ".java": "Java",46 ".kt": "Kotlin",47 ".go": "Go",48 ".rs": "Rust",49 ".c": "C",50 ".cpp": "C++",51 ".h": "C/C++ Header",52 ".hpp": "C++ Header",53 ".cs": "C#",54 ".php": "PHP",55 ".rb": "Ruby",56 ".swift": "Swift",57 }58 59 return mapping.get(suffix, "Unknown")60 61 62def extract_imports(text: str, path: str) -> list[str]:63 suffix = Path(path).suffix.lower()64 pattern = IMPORT_PATTERNS.get(suffix)65 if not pattern:66 return []67 68 matches = pattern.findall(text)69 return [m.strip() for m in matches if m.strip()]70 71 72def extract_todos(text: str) -> list[str]:73 todos: list[str] = []74 for line in text.splitlines():75 stripped = line.strip()76 upper = stripped.upper()77 if "TODO" in upper or "FIXME" in upper or "XXX" in upper:78 todos.append(stripped)79 return todos80 81 82def extract_python_symbols(text: str) -> tuple[list[str], list[str]]:83 classes = re.findall(r"^\s*class\s+([A-Za-z_][A-Za-z0-9_]*)", text, re.MULTILINE)84 functions = re.findall(r"^\s*def\s+([A-Za-z_][A-Za-z0-9_]*)", text, re.MULTILINE)85 return classes, functions86 87 88def extract_js_symbols(text: str) -> tuple[list[str], list[str]]:89 classes = re.findall(r"^\s*class\s+([A-Za-z_][A-Za-z0-9_]*)", text, re.MULTILINE)90 functions = re.findall(91 r"^\s*(?:function\s+([A-Za-z_][A-Za-z0-9_]*)|const\s+([A-Za-z_][A-Za-z0-9_]*)\s*=\s*\()",92 text,93 re.MULTILINE,94 )95 96 flat_functions: list[str] = []97 for fn_a, fn_b in functions:98 name = fn_a or fn_b99 if name:100 flat_functions.append(name)101 102 return classes, flat_functions103 104 105def parse_code_file(path: str) -> ParsedCodeInfo:106 file_path = Path(path)107 text = read_text(file_path)108 language = detect_language(path)109 110 imports = extract_imports(text, path)111 todos = extract_todos(text)112 113 classes: list[str] = []114 functions: list[str] = []115 116 if file_path.suffix.lower() == ".py":117 classes, functions = extract_python_symbols(text)118 elif file_path.suffix.lower() in {".js", ".jsx", ".ts", ".tsx"}:119 classes, functions = extract_js_symbols(text)120 121 return ParsedCodeInfo(122 path=path,123 language=language,124 line_count=len(text.splitlines()),125 imports=imports,126 todos=todos,127 classes=classes,128 functions=functions,129 )130 131 132def parse_repository_files(files: list[str], root_dir: str) -> list[ParsedCodeInfo]:133 """134 Parse a list of repository-relative file paths into structured code metadata.135 """136 root = Path(root_dir).expanduser().resolve()137 parsed: list[ParsedCodeInfo] = []138 139 for rel_path in files:140 full_path = root / rel_path141 if full_path.is_file():142 parsed.append(parse_code_file(str(full_path)))143 144 return parsed