Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
_redaction.py455 linesDownload Raw Back to middleware
1"""Shared redaction utilities for middleware components."""2 3from __future__ import annotations4 5import hashlib6import ipaddress7import operator8import re9from collections.abc import Callable, Sequence10from dataclasses import dataclass11from typing import Literal12from urllib.parse import urlparse13 14from typing_extensions import TypedDict15 16RedactionStrategy = Literal["block", "redact", "mask", "hash"]17"""Supported strategies for handling detected sensitive values."""18 19 20class PIIMatch(TypedDict):21    """Represents an individual match of sensitive data."""22 23    type: str24    value: str25    start: int26    end: int27 28 29class PIIDetectionError(Exception):30    """Raised when configured to block on detected sensitive values."""31 32    def __init__(self, pii_type: str, matches: Sequence[PIIMatch]) -> None:33        """Initialize the exception with match context.34 35        Args:36            pii_type: Name of the detected sensitive type.37            matches: All matches that were detected for that type.38        """39        self.pii_type = pii_type40        self.matches = list(matches)41        count = len(matches)42        msg = f"Detected {count} instance(s) of {pii_type} in text content"43        super().__init__(msg)44 45 46Detector = Callable[[str], list[PIIMatch]]47"""Callable signature for detectors that locate sensitive values."""48 49 50def detect_email(content: str) -> list[PIIMatch]:51    """Detect email addresses in content.52 53    Args:54        content: The text content to scan for email addresses.55 56    Returns:57        A list of detected email matches.58    """59    pattern = r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b"60    return [61        PIIMatch(62            type="email",63            value=match.group(),64            start=match.start(),65            end=match.end(),66        )67        for match in re.finditer(pattern, content)68    ]69 70 71def detect_credit_card(content: str) -> list[PIIMatch]:72    """Detect credit card numbers in content using Luhn validation.73 74    Args:75        content: The text content to scan for credit card numbers.76 77    Returns:78        A list of detected credit card matches.79    """80    pattern = r"\b\d{4}[\s-]?\d{4}[\s-]?\d{4}[\s-]?\d{4}\b"81    matches = []82 83    for match in re.finditer(pattern, content):84        card_number = match.group()85        if _passes_luhn(card_number):86            matches.append(87                PIIMatch(88                    type="credit_card",89                    value=card_number,90                    start=match.start(),91                    end=match.end(),92                )93            )94 95    return matches96 97 98def detect_ip(content: str) -> list[PIIMatch]:99    """Detect IPv4 or IPv6 addresses in content.100 101    Args:102        content: The text content to scan for IP addresses.103 104    Returns:105        A list of detected IP address matches.106    """107    matches: list[PIIMatch] = []108    ipv4_pattern = r"\b(?:[0-9]{1,3}\.){3}[0-9]{1,3}\b"109 110    for match in re.finditer(ipv4_pattern, content):111        ip_candidate = match.group()112        try:113            ipaddress.ip_address(ip_candidate)114        except ValueError:115            continue116        matches.append(117            PIIMatch(118                type="ip",119                value=ip_candidate,120                start=match.start(),121                end=match.end(),122            )123        )124 125    return matches126 127 128def detect_mac_address(content: str) -> list[PIIMatch]:129    """Detect MAC addresses in content.130 131    Args:132        content: The text content to scan for MAC addresses.133 134    Returns:135        A list of detected MAC address matches.136    """137    pattern = r"\b([0-9A-Fa-f]{2}[:-]){5}[0-9A-Fa-f]{2}\b"138    return [139        PIIMatch(140            type="mac_address",141            value=match.group(),142            start=match.start(),143            end=match.end(),144        )145        for match in re.finditer(pattern, content)146    ]147 148 149def detect_url(content: str) -> list[PIIMatch]:150    """Detect URLs in content using regex and stdlib validation.151 152    Args:153        content: The text content to scan for URLs.154 155    Returns:156        A list of detected URL matches.157    """158    matches: list[PIIMatch] = []159 160    # Pattern 1: URLs with scheme (http:// or https://)161    scheme_pattern = r"https?://[^\s<>\"{}|\\^`\[\]]+"162 163    for match in re.finditer(scheme_pattern, content):164        url = match.group()165        result = urlparse(url)166        if result.scheme in {"http", "https"} and result.netloc:167            matches.append(168                PIIMatch(169                    type="url",170                    value=url,171                    start=match.start(),172                    end=match.end(),173                )174            )175 176    # Pattern 2: URLs without scheme (www.example.com or example.com/path)177    # More conservative to avoid false positives178    bare_pattern = (179        r"\b(?:www\.)?[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?"180        r"(?:\.[a-zA-Z0-9](?:[a-zA-Z0-9-]{0,61}[a-zA-Z0-9])?)+(?:/[^\s]*)?"181    )182 183    for match in re.finditer(bare_pattern, content):184        start, end = match.start(), match.end()185        # Skip if already matched with scheme186        if any(m["start"] <= start < m["end"] or m["start"] < end <= m["end"] for m in matches):187            continue188 189        url = match.group()190        # Only accept if it has a path or starts with www191        # This reduces false positives like "example.com" in prose192        if "/" in url or url.startswith("www."):193            # Add scheme for validation (required for urlparse to work correctly)194            test_url = f"http://{url}"195            result = urlparse(test_url)196            if result.netloc and "." in result.netloc:197                matches.append(198                    PIIMatch(199                        type="url",200                        value=url,201                        start=start,202                        end=end,203                    )204                )205 206    return matches207 208 209BUILTIN_DETECTORS: dict[str, Detector] = {210    "email": detect_email,211    "credit_card": detect_credit_card,212    "ip": detect_ip,213    "mac_address": detect_mac_address,214    "url": detect_url,215}216"""Registry of built-in detectors keyed by type name."""217 218_CARD_NUMBER_MIN_DIGITS = 13219_CARD_NUMBER_MAX_DIGITS = 19220 221 222def _passes_luhn(card_number: str) -> bool:223    """Validate credit card number using the Luhn checksum."""224    digits = [int(d) for d in card_number if d.isdigit()]225    if not _CARD_NUMBER_MIN_DIGITS <= len(digits) <= _CARD_NUMBER_MAX_DIGITS:226        return False227 228    checksum = 0229    for index, digit in enumerate(reversed(digits)):230        value = digit231        if index % 2 == 1:232            value *= 2233            if value > 9:  # noqa: PLR2004234                value -= 9235        checksum += value236    return checksum % 10 == 0237 238 239def _apply_redact_strategy(content: str, matches: list[PIIMatch]) -> str:240    result = content241    for match in sorted(matches, key=operator.itemgetter("start"), reverse=True):242        replacement = f"[REDACTED_{match['type'].upper()}]"243        result = result[: match["start"]] + replacement + result[match["end"] :]244    return result245 246 247_UNMASKED_CHAR_NUMBER = 4248_IPV4_PARTS_NUMBER = 4249 250 251def _apply_mask_strategy(content: str, matches: list[PIIMatch]) -> str:252    result = content253    for match in sorted(matches, key=operator.itemgetter("start"), reverse=True):254        value = match["value"]255        pii_type = match["type"]256        if pii_type == "email":257            parts = value.split("@")258            if len(parts) == 2:  # noqa: PLR2004259                domain_parts = parts[1].split(".")260                masked = (261                    f"{parts[0]}@****.{domain_parts[-1]}"262                    if len(domain_parts) > 1263                    else f"{parts[0]}@****"264                )265            else:266                masked = "****"267        elif pii_type == "credit_card":268            digits_only = "".join(c for c in value if c.isdigit())269            separator = "-" if "-" in value else " " if " " in value else ""270            if separator:271                masked = (272                    f"****{separator}****{separator}****{separator}"273                    f"{digits_only[-_UNMASKED_CHAR_NUMBER:]}"274                )275            else:276                masked = f"************{digits_only[-_UNMASKED_CHAR_NUMBER:]}"277        elif pii_type == "ip":278            octets = value.split(".")279            masked = f"*.*.*.{octets[-1]}" if len(octets) == _IPV4_PARTS_NUMBER else "****"280        elif pii_type == "mac_address":281            separator = ":" if ":" in value else "-"282            masked = (283                f"**{separator}**{separator}**{separator}**{separator}**{separator}{value[-2:]}"284            )285        elif pii_type == "url":286            masked = "[MASKED_URL]"287        else:288            masked = (289                f"****{value[-_UNMASKED_CHAR_NUMBER:]}"290                if len(value) > _UNMASKED_CHAR_NUMBER291                else "****"292            )293        result = result[: match["start"]] + masked + result[match["end"] :]294    return result295 296 297def _apply_hash_strategy(content: str, matches: list[PIIMatch]) -> str:298    result = content299    for match in sorted(matches, key=operator.itemgetter("start"), reverse=True):300        digest = hashlib.sha256(match["value"].encode()).hexdigest()[:8]301        replacement = f"<{match['type']}_hash:{digest}>"302        result = result[: match["start"]] + replacement + result[match["end"] :]303    return result304 305 306def apply_strategy(307    content: str,308    matches: list[PIIMatch],309    strategy: RedactionStrategy,310) -> str:311    """Apply the configured strategy to matches within content.312 313    Args:314        content: The content to apply strategy to.315        matches: List of detected PII matches.316        strategy: The redaction strategy to apply.317 318    Returns:319        The content with the strategy applied.320 321    Raises:322        PIIDetectionError: If the strategy is `'block'` and matches are found.323        ValueError: If the strategy is unknown.324    """325    if not matches:326        return content327    if strategy == "redact":328        return _apply_redact_strategy(content, matches)329    if strategy == "mask":330        return _apply_mask_strategy(content, matches)331    if strategy == "hash":332        return _apply_hash_strategy(content, matches)333    if strategy == "block":334        raise PIIDetectionError(matches[0]["type"], matches)335    msg = f"Unknown redaction strategy: {strategy}"  # type: ignore[unreachable]336    raise ValueError(msg)337 338 339def resolve_detector(pii_type: str, detector: Detector | str | None) -> Detector:340    """Return a callable detector for the given configuration.341 342    Args:343        pii_type: The PII type name.344        detector: Optional custom detector or regex pattern. If `None`, a built-in detector345            for the given PII type will be used.346 347    Returns:348        The resolved detector.349 350    Raises:351        ValueError: If an unknown PII type is specified without a custom detector or regex.352    """353    if detector is None:354        if pii_type not in BUILTIN_DETECTORS:355            msg = (356                f"Unknown PII type: {pii_type}. "357                f"Must be one of {list(BUILTIN_DETECTORS.keys())} or provide a custom detector."358            )359            raise ValueError(msg)360        return BUILTIN_DETECTORS[pii_type]361    if isinstance(detector, str):362        pattern = re.compile(detector)363 364        def regex_detector(content: str) -> list[PIIMatch]:365            return [366                PIIMatch(367                    type=pii_type,368                    value=match.group(),369                    start=match.start(),370                    end=match.end(),371                )372                for match in pattern.finditer(content)373            ]374 375        return regex_detector376 377    # Wrap the custom callable to normalize its output.378    # Custom detectors may return dicts with "text" instead of "value"379    # and may omit "type".  Map them to proper PIIMatch objects so that380    # downstream strategies (hash, mask) can access match["value"].381    raw_detector = detector382 383    def _normalizing_detector(content: str) -> list[PIIMatch]:384        return [385            PIIMatch(386                type=m.get("type", pii_type),387                value=m.get("value", m.get("text", "")),388                start=m["start"],389                end=m["end"],390            )391            for m in raw_detector(content)392        ]393 394    return _normalizing_detector395 396 397@dataclass(frozen=True)398class RedactionRule:399    """Configuration for handling a single PII type."""400 401    pii_type: str402    strategy: RedactionStrategy = "redact"403    detector: Detector | str | None = None404 405    def resolve(self) -> ResolvedRedactionRule:406        """Resolve runtime detector and return an immutable rule.407 408        Returns:409            The resolved redaction rule.410        """411        resolved_detector = resolve_detector(self.pii_type, self.detector)412        return ResolvedRedactionRule(413            pii_type=self.pii_type,414            strategy=self.strategy,415            detector=resolved_detector,416        )417 418 419@dataclass(frozen=True)420class ResolvedRedactionRule:421    """Resolved redaction rule ready for execution."""422 423    pii_type: str424    strategy: RedactionStrategy425    detector: Detector426 427    def apply(self, content: str) -> tuple[str, list[PIIMatch]]:428        """Apply this rule to content, returning new content and matches.429 430        Args:431            content: The text content to scan and redact.432 433        Returns:434            A tuple of (updated content, list of detected matches).435        """436        matches = self.detector(content)437        if not matches:438            return content, []439        updated = apply_strategy(content, matches, self.strategy)440        return updated, matches441 442 443__all__ = [444    "PIIDetectionError",445    "PIIMatch",446    "RedactionRule",447    "ResolvedRedactionRule",448    "apply_strategy",449    "detect_credit_card",450    "detect_email",451    "detect_ip",452    "detect_mac_address",453    "detect_url",454]455 
codekingpro/portable-devtools · Team Ai