Team Ai
Apppublic

Sejal1908/email-classification-api

sourceHugging Facemitupdated 1y agoView on Hugging Face
0likes
utils.py72 linesDownload Raw Back to root
1import re2 3# Strict patterns for PII/PCI4PII_PATTERNS = {5    "full_name":  r'\b(?:Mr\.|Mrs\.|Ms\.|Dr\.)?\s*[A-Z][a-z]+(?:\s[A-Z][a-z]+)+\b',6    "email": r"\b[\w\.-]+@[\w\.-]+\.\w{2,4}\b",7    "phone_number": r"\+?\d{1,4}[-\s]?\(?\d{1,4}\)?[-\s]?\d{1,4}[-\s]?\d{1,4}[-\s]?\d{1,4}",8    "dob": r"\b\d{2}[/-]\d{2}[/-]\d{4}\b",  # 12/25/19909    "aadhar_num": r"\b\d{12}\b",10    "credit_debit_no": r"\b(?:\d[ -]*?){13,16}\b",11    "cvv_no": r"\b\d{3}\b",12    # expiry only MM/YY, not DOB13    "expiry_no": r"\b(0[1-9]|1[0-2])[/-]\d{2}\b",14}15 16 17def mask_pii_entities(text: str) -> dict:18    """19    Detects and masks PII/PCI fields in `text`, returning:20      {21        "masked_email": "...",22        "masked_entities": [23           {"position":[s,e], "classification":"dob", "entity":"12/25/1990"}, ...24        ]25      }26    Overlapping matches are resolved in favor of the longest match first.27    """28    # 1) Gather all regex matches29    raw_matches = []30    for label, pattern in PII_PATTERNS.items():31        for m in re.finditer(pattern, text):32            raw_matches.append(33                {34                    "classification": label,35                    "entity": m.group(),36                    "position": [m.start(), m.end()],37                    "length": m.end() - m.start(),38                }39            )40 41    # 2) Sort by descending match length (so dob > expiry_no)42    raw_matches.sort(key=lambda x: x["length"], reverse=True)43 44    # 3) Filter out overlaps45    used_positions = set()46    final_matches = []47    for m in raw_matches:48        s, e = m["position"]49        if not any(pos in used_positions for pos in range(s, e)):50            final_matches.append(m)51            used_positions.update(range(s, e))52 53    # 4) Mask in one pass (from left to right), adjusting offsets54    masked = text55    offset = 056    for m in sorted(final_matches, key=lambda x: x["position"][0]):57        s, e = m["position"]58        tag = f"[{m['classification']}]"59        masked = masked[: s + offset] + tag + masked[e + offset :]60        offset += len(tag) - (e - s)61 62    # 5) Build masked_entities list (drop the internal 'length' key)63    masked_entities = [64        {65            "position": m["position"],66            "classification": m["classification"],67            "entity": m["entity"],68        }69        for m in final_matches70    ]71 72    return {"masked_email": masked, "masked_entities": masked_entities}