24f3001764/llm_code_deployment
0
1import re2import logging3from pathlib import Path4from typing import List, Tuple5 6logger = logging.getLogger(__name__)7 8 9class SecurityScanner:10 """Scan code for potential secrets and sensitive information"""11 12 # Common patterns for secrets13 SECRET_PATTERNS = [14 (r'(?i)(api[_-]?key|apikey)\s*[:=]\s*["\']([a-zA-Z0-9_\-]{20,})["\']', 'API Key'),15 (r'(?i)(secret[_-]?key|secretkey)\s*[:=]\s*["\']([a-zA-Z0-9_\-]{20,})["\']', 'Secret Key'),16 (r'(?i)(password|passwd|pwd)\s*[:=]\s*["\']([^"\']{8,})["\']', 'Password'),17 (r'(?i)(token)\s*[:=]\s*["\']([a-zA-Z0-9_\-]{20,})["\']', 'Token'),18 (r'(?i)(github[_-]?token)\s*[:=]\s*["\']([a-zA-Z0-9_\-]{20,})["\']', 'GitHub Token'),19 (r'(?i)(openai[_-]?api[_-]?key)\s*[:=]\s*["\']([a-zA-Z0-9_\-]{20,})["\']', 'OpenAI API Key'),20 (r'sk-[a-zA-Z0-9]{20,}', 'OpenAI API Key (sk- prefix)'),21 (r'ghp_[a-zA-Z0-9]{36,}', 'GitHub Personal Access Token'),22 (r'gho_[a-zA-Z0-9]{36,}', 'GitHub OAuth Token'),23 (r'ghs_[a-zA-Z0-9]{36,}', 'GitHub App Token'),24 (r'(?i)bearer\s+[a-zA-Z0-9_\-\.]{20,}', 'Bearer Token'),25 (r'(?i)(aws[_-]?access[_-]?key[_-]?id)\s*[:=]\s*["\']([A-Z0-9]{20})["\']', 'AWS Access Key'),26 (r'(?i)(aws[_-]?secret[_-]?access[_-]?key)\s*[:=]\s*["\']([a-zA-Z0-9/+=]{40})["\']', 'AWS Secret Key'),27 (r'-----BEGIN\s+(?:RSA\s+)?PRIVATE\s+KEY-----', 'Private Key'),28 (r'(?i)(database[_-]?url|db[_-]?url)\s*[:=]\s*["\']([^"\']+)["\']', 'Database URL'),29 ]30 31 # Whitelist patterns (things that look like secrets but aren't)32 WHITELIST_PATTERNS = [33 r'example\.com',34 r'your-.*-here',35 r'placeholder',36 r'dummy',37 r'test[_-]?key',38 r'fake[_-]?token',39 r'xxx+',40 r'\*\*\*+',41 ]42 43 def scan_file(self, file_path: Path) -> List[Tuple[str, str, int]]:44 """45 Scan a file for potential secrets46 Returns: List of (secret_type, matched_text, line_number)47 """48 findings = []49 50 try:51 with open(file_path, 'r', encoding='utf-8', errors='ignore') as f:52 lines = f.readlines()53 54 for line_num, line in enumerate(lines, 1):55 # Skip comments56 if line.strip().startswith(('#', '//', '/*', '*')):57 continue58 59 for pattern, secret_type in self.SECRET_PATTERNS:60 matches = re.finditer(pattern, line)61 for match in matches:62 matched_text = match.group(0)63 64 # Check if it's whitelisted65 if not self._is_whitelisted(matched_text):66 findings.append((secret_type, matched_text, line_num))67 68 except Exception as e:69 logger.warning(f"Error scanning {file_path}: {e}")70 71 return findings72 73 def _is_whitelisted(self, text: str) -> bool:74 """Check if text matches whitelist patterns"""75 for pattern in self.WHITELIST_PATTERNS:76 if re.search(pattern, text, re.IGNORECASE):77 return True78 return False79 80 def scan_directory(self, directory: Path) -> dict:81 """82 Scan all files in a directory83 Returns: Dict mapping file paths to findings84 """85 results = {}86 87 # File extensions to scan88 extensions = ['.html', '.js', '.css', '.py', '.json', '.yaml', '.yml', '.env', '.txt', '.md']89 90 for file_path in directory.rglob('*'):91 if file_path.is_file() and file_path.suffix in extensions:92 findings = self.scan_file(file_path)93 if findings:94 results[str(file_path.relative_to(directory))] = findings95 96 return results97 98 def scan_and_report(self, directory: Path) -> bool:99 """100 Scan directory and log findings101 Returns: True if no secrets found, False if secrets detected102 """103 logger.info(f"Scanning {directory} for secrets...")104 results = self.scan_directory(directory)105 106 if not results:107 logger.info("✓ No secrets detected")108 return True109 110 logger.warning(f"⚠ Found potential secrets in {len(results)} file(s):")111 for file_path, findings in results.items():112 logger.warning(f" {file_path}:")113 for secret_type, matched_text, line_num in findings:114 # Mask the secret for logging115 masked = self._mask_secret(matched_text)116 logger.warning(f" Line {line_num}: {secret_type} - {masked}")117 118 return False119 120 def _mask_secret(self, text: str) -> str:121 """Mask secret for safe logging"""122 if len(text) <= 8:123 return '*' * len(text)124 return text[:4] + '*' * (len(text) - 8) + text[-4:]125 126 def sanitize_file(self, file_path: Path) -> bool:127 """128 Remove detected secrets from a file129 Returns: True if file was modified, False otherwise130 """131 findings = self.scan_file(file_path)132 if not findings:133 return False134 135 try:136 with open(file_path, 'r', encoding='utf-8') as f:137 content = f.read()138 139 original_content = content140 141 # Replace secrets with placeholders142 for secret_type, matched_text, _ in findings:143 if not self._is_whitelisted(matched_text):144 placeholder = f"[REDACTED_{secret_type.upper().replace(' ', '_')}]"145 content = content.replace(matched_text, placeholder)146 147 if content != original_content:148 with open(file_path, 'w', encoding='utf-8') as f:149 f.write(content)150 logger.info(f"Sanitized {file_path}")151 return True152 153 except Exception as e:154 logger.error(f"Error sanitizing {file_path}: {e}")155 156 return False157 