Team Ai
Apppublic

Mohanrajdeena/Intelligent-document-processing

sourceHugging Facemitupdated 1mo agoView on Hugging Face
0likes
app.py1405 linesDownload Raw Back to root
1import gradio as gr2import spaces3import cv24import numpy as np5import easyocr6import torch7import re8import json9import time10import tempfile11 12from pathlib import Path13from datetime import datetime14 15 16# ============================================================17# CONFIGURATION18# ============================================================19 20SUPPORTED_TYPES = [21    "Invoice",22    "Resume",23    "ID Card"24]25 26 27SKILL_DICTIONARY = [28    "Python",29    "SQL",30    "MySQL",31    "PostgreSQL",32    "Power BI",33    "Excel",34    "Power Query",35    "Tableau",36    "Machine Learning",37    "Deep Learning",38    "Pandas",39    "NumPy",40    "Matplotlib",41    "Seaborn",42    "Streamlit",43    "TensorFlow",44    "PyTorch",45    "Scikit-learn",46    "Java",47    "C++",48    "C",49    "AWS",50    "Azure",51    "Git",52    "GitHub",53    "SAP",54    "Oracle"55]56 57 58# ============================================================59# IMAGE UTILITIES60# ============================================================61 62def read_image(image_path):63 64    if not image_path:65        raise ValueError(66            "Please upload a document image."67        )68 69    image = cv2.imread(70        str(image_path)71    )72 73    if image is None:74        raise ValueError(75            "Unable to read the uploaded image."76        )77 78    return image79 80 81def resize_image(82    image,83    max_width=160084):85 86    height, width = image.shape[:2]87 88    if width <= max_width:89        return image90 91    scale = max_width / width92 93    new_height = int(94        height * scale95    )96 97    return cv2.resize(98        image,99        (100            max_width,101            new_height102        ),103        interpolation=cv2.INTER_AREA104    )105 106 107# ============================================================108# DOCUMENT PREPROCESSING109# ============================================================110 111def preprocess_document(image):112 113    image = resize_image(114        image,115        max_width=1600116    )117 118    gray = cv2.cvtColor(119        image,120        cv2.COLOR_BGR2GRAY121    )122 123    # CLAHE enhancement124    clahe = cv2.createCLAHE(125        clipLimit=2.0,126        tileGridSize=(8, 8)127    )128 129    enhanced = clahe.apply(130        gray131    )132 133    # Gaussian denoising134    denoised = cv2.GaussianBlur(135        enhanced,136        (3, 3),137        0138    )139 140    # Otsu thresholding141    thresholded = cv2.threshold(142        denoised,143        0,144        255,145        cv2.THRESH_BINARY +146        cv2.THRESH_OTSU147    )[1]148 149    return thresholded150 151 152# ============================================================153# OCR154# ============================================================155 156def create_ocr_reader():157 158    # ZeroGPU provides CUDA while the159    # @spaces.GPU function is executing.160 161    gpu_available = torch.cuda.is_available()162 163    print(164        f"Initializing EasyOCR | GPU: {gpu_available}"165    )166 167    reader = easyocr.Reader(168        ["en"],169        gpu=gpu_available,170        verbose=False171    )172 173    return reader174 175 176def perform_ocr(177    image,178    reader179):180 181    results = reader.readtext(182        image,183        detail=1,184        paragraph=False185    )186 187    text_lines = []188    confidence_scores = []189 190    for result in results:191 192        if len(result) >= 3:193 194            text_lines.append(195                str(result[1])196            )197 198            confidence_scores.append(199                float(result[2])200            )201 202    full_text = "\n".join(203        text_lines204    )205 206    average_confidence = (207        sum(confidence_scores)208        /209        len(confidence_scores)210        if confidence_scores211        else 0.0212    )213 214    return {215        "text": full_text,216        "confidence": average_confidence,217        "results": results218    }219 220 221# ============================================================222# TEXT CLEANING223# ============================================================224 225def clean_text(text):226 227    text = re.sub(228        r"[ \t]+",229        " ",230        text231    )232 233    text = re.sub(234        r"\n{2,}",235        "\n",236        text237    )238 239    return text.strip()240 241 242# ============================================================243# GENERIC HELPERS244# ============================================================245 246def unique(values):247 248    return list(249        dict.fromkeys(values)250    )251 252 253def extract_emails(text):254 255    pattern = (256        r"\b[A-Za-z0-9._%+-]+"257        r"@[A-Za-z0-9.-]+\."258        r"[A-Za-z]{2,}\b"259    )260 261    return unique(262        re.findall(263            pattern,264            text265        )266    )267 268 269def extract_phone_numbers(text):270 271    pattern = (272        r"(?<!\d)"273        r"(?:\+91[\s-]?)?"274        r"[6-9]\d{9}"275        r"(?!\d)"276    )277 278    return unique(279        re.findall(280            pattern,281            text282        )283    )284 285 286def extract_dates(text):287 288    pattern = (289        r"\b(?:"290        r"\d{1,2}[/-]\d{1,2}[/-]\d{2,4}"291        r"|"292        r"\d{1,2}[/-]"293        r"[A-Za-z]{3,9}[/-]?"294        r"\d{2,4}"295        r"|"296        r"[A-Za-z]{3,9}\s+"297        r"\d{1,2},?\s+\d{2,4}"298        r")\b"299    )300 301    return unique(302        re.findall(303            pattern,304            text,305            flags=re.IGNORECASE306        )307    )308 309 310def extract_amounts(text):311 312    pattern = (313        r"(?:โ‚น|Rs\.?|INR|\$|โ‚ฌ|ยฃ)\s*"314        r"\d+(?:,\d{3})*(?:\.\d{1,2})?"315    )316 317    return unique(318        re.findall(319            pattern,320            text,321            flags=re.IGNORECASE322        )323    )324 325 326# ============================================================327# INVOICE EXTRACTION328# ============================================================329 330def extract_invoice_number(text):331 332    patterns = [333 334        (335            r"(?:invoice\s*"336            r"(?:no|number|#)"337            r"\s*[:\-]?\s*)"338            r"([A-Z0-9\/\-_]+)"339        ),340 341        (342            r"(?:inv\s*"343            r"(?:no|number|#)"344            r"\s*[:\-]?\s*)"345            r"([A-Z0-9\/\-_]+)"346        )347    ]348 349    for pattern in patterns:350 351        match = re.search(352            pattern,353            text,354            flags=re.IGNORECASE355        )356 357        if match:358 359            return (360                match.group(1)361                .strip()362            )363 364    return None365 366 367def extract_gstin(text):368 369    pattern = (370        r"\b"371        r"\d{2}"372        r"[A-Z]{5}"373        r"\d{4}"374        r"[A-Z]"375        r"\d"376        r"[A-Z0-9]"377        r"\b"378    )379 380    return unique(381        re.findall(382            pattern,383            text.upper()384        )385    )386 387 388def extract_invoice_entities(text):389 390    return {391 392        "Invoice Number":393            extract_invoice_number(text),394 395        "Invoice Date":396            extract_dates(text),397 398        "GSTIN":399            extract_gstin(text),400 401        "Amounts":402            extract_amounts(text),403 404        "Email":405            extract_emails(text),406 407        "Phone Number":408            extract_phone_numbers(text)409    }410 411 412# ============================================================413# RESUME EXTRACTION414# ============================================================415 416def extract_name_from_resume(text):417 418    lines = [419 420        line.strip()421 422        for line in text.split("\n")423 424        if line.strip()425    ]426 427    if lines:428 429        candidate = lines[0]430 431        if (432            len(candidate.split()) <= 5433            and len(candidate) <= 60434        ):435 436            return candidate437 438    return None439 440 441def extract_skills(text):442 443    text_lower = text.lower()444 445    return [446 447        skill448 449        for skill in SKILL_DICTIONARY450 451        if skill.lower()452        in text_lower453    ]454 455 456def extract_resume_entities(text):457 458    return {459 460        "Name":461            extract_name_from_resume(text),462 463        "Email":464            extract_emails(text),465 466        "Phone Number":467            extract_phone_numbers(text),468 469        "Skills":470            extract_skills(text),471 472        "Dates":473            extract_dates(text)474    }475 476 477# ============================================================478# ID CARD EXTRACTION479# ============================================================480 481def extract_aadhaar_numbers(text):482 483    pattern = (484        r"\b\d{4}\s\d{4}\s\d{4}\b"485    )486 487    return unique(488        re.findall(489            pattern,490            text491        )492    )493 494 495def extract_id_card_entities(text):496 497    return {498 499        "Name":500            extract_name_from_resume(text),501 502        "ID Number":503            extract_aadhaar_numbers(text),504 505        "Date":506            extract_dates(text),507 508        "Phone Number":509            extract_phone_numbers(text)510    }511 512 513# ============================================================514# DOCUMENT ROUTER515# ============================================================516 517def extract_entities(518    text,519    document_type520):521 522    if document_type == "Invoice":523 524        return extract_invoice_entities(525            text526        )527 528    if document_type == "Resume":529 530        return extract_resume_entities(531            text532        )533 534    if document_type == "ID Card":535 536        return extract_id_card_entities(537            text538        )539 540    return {}541 542 543# ============================================================544# OUTPUT HELPERS545# ============================================================546 547def count_detected_entities(548    entities549):550 551    count = 0552 553    for value in entities.values():554 555        if value is None:556            continue557 558        if isinstance(559            value,560            list561        ):562 563            count += len(value)564 565        elif str(value).strip():566 567            count += 1568 569    return count570 571 572def format_entities(573    entities574):575 576    formatted = {}577 578    for key, value in entities.items():579 580        if value is None:581 582            formatted[key] = (583                "Not detected"584            )585 586        elif isinstance(587            value,588            list589        ):590 591            formatted[key] = (592 593                ", ".join(594                    str(x)595                    for x in value596                )597 598                if value599 600                else "Not detected"601            )602 603        else:604 605            formatted[key] = (606                str(value)607            )608 609    return formatted610 611 612def create_json_file(data):613 614    temp = tempfile.NamedTemporaryFile(615        mode="w",616        suffix=".json",617        prefix="document_results_",618        delete=False,619        encoding="utf-8"620    )621 622    json.dump(623        data,624        temp,625        indent=4,626        ensure_ascii=False627    )628 629    temp.close()630 631    return temp.name632 633 634# ============================================================635# MAIN PROCESSING FUNCTION636# ============================================================637 638@spaces.GPU(duration=120)639def process_document(640    document_type,641    image_path642):643 644    start_time = time.perf_counter()645 646    if not image_path:647 648        raise gr.Error(649            "Please upload an invoice, "650            "resume, or ID card image."651        )652 653    if document_type not in SUPPORTED_TYPES:654 655        raise gr.Error(656            "Please select a valid "657            "document type."658        )659 660    try:661 662        # ----------------------------------------------------663        # STEP 1: LOAD IMAGE664        # ----------------------------------------------------665 666        original_bgr = read_image(667            image_path668        )669 670        original_rgb = cv2.cvtColor(671            original_bgr,672            cv2.COLOR_BGR2RGB673        )674 675        # ----------------------------------------------------676        # STEP 2: PREPROCESS677        # ----------------------------------------------------678 679        processed_image = (680            preprocess_document(681                original_bgr682            )683        )684 685        # ----------------------------------------------------686        # STEP 3: LOAD EASYOCR687        # ----------------------------------------------------688 689        # IMPORTANT:690        # Reader is created inside the GPU-decorated691        # function so ZeroGPU can provide CUDA.692 693        reader = create_ocr_reader()694 695        # ----------------------------------------------------696        # STEP 4: OCR697        # ----------------------------------------------------698 699        ocr_output = perform_ocr(700            processed_image,701            reader702        )703 704        raw_text = (705            ocr_output["text"]706        )707 708        confidence = float(709            ocr_output["confidence"]710        )711 712        # ----------------------------------------------------713        # STEP 5: CLEAN TEXT714        # ----------------------------------------------------715 716        cleaned_text = clean_text(717            raw_text718        )719 720        # ----------------------------------------------------721        # STEP 6: ENTITY EXTRACTION722        # ----------------------------------------------------723 724        entities = extract_entities(725            cleaned_text,726            document_type727        )728 729        formatted_entities = (730            format_entities(731                entities732            )733        )734 735        detected_fields = (736            count_detected_entities(737                entities738            )739        )740 741        # ----------------------------------------------------742        # STEP 7: PROCESSING TIME743        # ----------------------------------------------------744 745        processing_time = (746            time.perf_counter()747            - start_time748        )749 750        # ----------------------------------------------------751        # STEP 8: FINAL JSON752        # ----------------------------------------------------753 754        final_output = {755 756            "document_type":757                document_type,758 759            "file_name":760                Path(761                    image_path762                ).name,763 764            "processing_timestamp":765                datetime.now().isoformat(),766 767            "ocr_confidence":768                round(769                    confidence,770                    4771                ),772 773            "ocr_confidence_percent":774                round(775                    confidence * 100,776                    2777                ),778 779            "processing_time_seconds":780                round(781                    processing_time,782                    3783                ),784 785            "detected_fields":786                detected_fields,787 788            "entities":789                formatted_entities,790 791            "ocr_text":792                raw_text,793 794            "cleaned_text":795                cleaned_text796        }797 798        # ----------------------------------------------------799        # STEP 9: CREATE DOWNLOADABLE JSON800        # ----------------------------------------------------801 802        json_file = create_json_file(803            final_output804        )805 806        # ----------------------------------------------------807        # STEP 10: SUMMARY808        # ----------------------------------------------------809 810        confidence_text = (811            f"{confidence * 100:.1f}%"812        )813 814        summary = f"""815### โœ… Processing Complete816 817| Metric | Result |818|---|---|819| ๐Ÿ“„ Document Type | **{document_type}** |820| ๐ŸŽฏ OCR Confidence | **{confidence_text}** |821| ๐Ÿ”Ž Fields Detected | **{detected_fields}** |822| ๐Ÿ“ OCR Characters | **{len(raw_text)}** |823| โšก Processing Time | **{processing_time:.2f} seconds** |824 825**Pipeline:**  826Preprocessing โ†’ EasyOCR โ†’ Text Cleaning โ†’ Entity Extraction827"""828 829        return (830 831            original_rgb,832 833            processed_image,834 835            formatted_entities,836 837            raw_text,838 839            cleaned_text,840 841            summary,842 843            json.dumps(844                final_output,845                indent=4,846                ensure_ascii=False847            ),848 849            json_file850        )851 852    except gr.Error:853 854        raise855 856    except Exception as error:857 858        print(859            "Processing error:",860            repr(error)861        )862 863        raise gr.Error(864            f"Processing failed: {error}"865        )866 867 868# ============================================================869# PROFESSIONAL UI CSS870# ============================================================871 872CSS = """873 874/* ================================875   PAGE876================================ */877 878body {879 880    background:881        linear-gradient(882            135deg,883            #f8faff 0%,884            #f3f0ff 50%,885            #eef7ff 100%886        ) !important;887}888 889 890/* ================================891   MAIN CONTAINER892================================ */893 894.gradio-container {895 896    max-width:897        1450px !important;898 899    margin:900        auto !important;901 902    padding:903        20px !important;904}905 906 907/* ================================908   HERO909================================ */910 911.hero {912 913    background:914        linear-gradient(915            135deg,916            #2563eb,917            #7c3aed918        );919 920    color:921        white;922 923    border-radius:924        20px;925 926    padding:927        30px 35px;928 929    margin-bottom:930        20px;931 932    box-shadow:933        0 12px 35px934        rgba(935            37,936            99,937            235,938            0.20939        );940}941 942.hero h1 {943 944    color:945        white !important;946 947    margin:948        0 !important;949 950    font-size:951        2.3rem !important;952 953    font-weight:954        800 !important;955}956 957.hero p {958 959    color:960        rgba(961            255,962            255,963            255,964            0.95965        ) !important;966 967    margin-top:968        10px !important;969 970    font-size:971        1.05rem !important;972}973 974 975/* ================================976   SECTION HEADINGS977================================ */978 979.section-title {980 981    font-size:982        1.2rem;983 984    font-weight:985        800;986 987    color:988        #1e293b !important;989}990 991 992/* ================================993   CARDS994================================ */995 996.card {997 998    background:999        white;1000 1001    border:1002        1px solid #e5e7eb;1003 1004    border-radius:1005        16px;1006 1007    padding:1008        18px;1009 1010    box-shadow:1011        0 6px 20px1012        rgba(1013            15,1014            23,1015            42,1016            0.061017        );1018}1019 1020 1021/* ================================1022   LABELS1023================================ */1024 1025label {1026 1027    color:1028        #1e293b !important;1029 1030    font-weight:1031        700 !important;1032}1033 1034 1035/* ================================1036   INPUTS1037================================ */1038 1039textarea,1040input {1041 1042    color:1043        #111827 !important;1044 1045    background:1046        #ffffff !important;1047}1048 1049 1050/* ================================1051   BUTTON1052================================ */1053 1054button.primary {1055 1056    background:1057        linear-gradient(1058            135deg,1059            #2563eb,1060            #7c3aed1061        ) !important;1062 1063    color:1064        white !important;1065 1066    border:1067        none !important;1068 1069    font-weight:1070        800 !important;1071 1072    border-radius:1073        10px !important;1074}1075 1076 1077/* ================================1078   OUTPUT1079================================ */1080 1081.output-markdown {1082 1083    color:1084        #1e293b !important;1085}1086 1087 1088/* ================================1089   FOOTER1090================================ */1091 1092footer {1093 1094    display:1095        none !important;1096}1097 1098"""1099 1100 1101# ============================================================1102# GRADIO APPLICATION1103# ============================================================1104 1105with gr.Blocks(1106    title="Intelligent Document Processing"1107) as demo:1108 1109    # --------------------------------------------------------1110    # HERO1111    # --------------------------------------------------------1112 1113    gr.HTML(1114        """1115        <div class="hero">1116 1117            <h1>1118                ๐Ÿ“„ Intelligent Document Processing1119            </h1>1120 1121            <p>1122                AI-powered OCR, text cleaning and1123                structured information extraction1124                for invoices, resumes and ID cards.1125            </p>1126 1127        </div>1128        """1129    )1130 1131 1132    # --------------------------------------------------------1133    # INPUT / PREVIEW1134    # --------------------------------------------------------1135 1136    with gr.Row():1137 1138        # ====================================================1139        # LEFT COLUMN1140        # ====================================================1141 1142        with gr.Column(1143            scale=1,1144            elem_classes=["card"]1145        ):1146 1147            gr.Markdown(1148                "### โš™๏ธ Document Configuration",1149                elem_classes=["section-title"]1150            )1151 1152            document_type = gr.Dropdown(1153 1154                choices=SUPPORTED_TYPES,1155 1156                value="Invoice",1157 1158                label="Document Type",1159 1160                info=(1161                    "Select the type of "1162                    "document you are uploading."1163                )1164            )1165 1166 1167            image_input = gr.Image(1168 1169                label="๐Ÿ“ค Upload Document",1170 1171                type="filepath",1172 1173                sources=["upload"],1174 1175                height=3501176            )1177 1178 1179            process_button = gr.Button(1180 1181                "๐Ÿš€ Extract Information",1182 1183                variant="primary",1184 1185                size="lg"1186            )1187 1188 1189            gr.Markdown(1190                """1191**Supported formats:** JPG, JPEG, PNG1192 1193**Supported documents:**1194 1195๐Ÿงพ Invoice  1196๐Ÿ“„ Resume  1197๐Ÿชช ID Card1198 1199**Processing pipeline:**1200 

Showing the first 1,200 of 1405 lines. Download the file for the rest.