Team Ai
Apppublic

AITestingWorkSpace/DocumentAnalysis

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
app.py490 linesDownload Raw Back to root
1import logging2import os3import docx4import PyPDF25from docx.shared import RGBColor, Pt6from io import BytesIO, IOBase7import tempfile8import re9import datetime10import torch11 12import gradio as gr13from transformers import AutoModelForCausalLM, AutoTokenizer14import huggingface_hub15 16###############################################################################17# 1) Logging Configuration18###############################################################################19logging.basicConfig(20    level=logging.INFO,  # or logging.DEBUG for more verbose logs21    format="%(asctime)s [%(levelname)s] %(name)s - %(message)s"22)23logger = logging.getLogger("LLM-Legal-App")24 25###############################################################################26# 2) Initialize Hugging Face Model27###############################################################################28def initialize_model():29    """Initialize the phi-2 model and tokenizer from HuggingFace."""30    logger.info("Initializing phi-2 model and tokenizer...")31    try:32        # Access token might be needed for some models33        # token = huggingface_hub.get_token()34        35        model_name = "microsoft/phi-2"36        tokenizer = AutoTokenizer.from_pretrained(model_name)37        model = AutoModelForCausalLM.from_pretrained(38            model_name,39            torch_dtype=torch.float16,40            device_map="auto",41            trust_remote_code=True42        )43        logger.info("Successfully initialized phi-2 model and tokenizer.")44        return model, tokenizer45    except Exception as e:46        logger.exception("Error initializing Hugging Face model.")47        raise ValueError(f"Failed to initialize model: {e}")48 49# Initialize model and tokenizer50model, tokenizer = initialize_model()51 52###############################################################################53# 3) LLM Utility Functions (Generation & Review)54###############################################################################55def generate_with_model(prompt, max_length=1400, temperature=0.3):56    """Generate text using the Hugging Face model."""57    logger.info("Generating text with phi-2 model.")58    59    try:60        inputs = tokenizer(prompt, return_tensors="pt").to(model.device)61        62        # Generate with parameters similar to the original OpenAI call63        generation_config = {64            "max_new_tokens": max_length,65            "temperature": temperature,66            "top_p": 0.9,67            "do_sample": temperature > 0,68            "pad_token_id": tokenizer.eos_token_id69        }70        71        with torch.no_grad():72            outputs = model.generate(**inputs, **generation_config)73        74        response = tokenizer.decode(outputs[0], skip_special_tokens=True)75        76        # Remove the prompt from the response77        if response.startswith(prompt):78            response = response[len(prompt):].strip()79            80        logger.info("Text generation complete.")81        return response82    83    except Exception as e:84        logger.exception("Error during text generation.")85        return f"Error generating text: {e}"86 87def generate_legal_document(doc_type, party_a, party_b, context, country):88    """89    Uses DocumentCogito to generate a legal document. Returns the document text.90    """91    logger.info(f"Starting generation for doc_type={doc_type!r}.")92    # Fill placeholders if fields are missing93    party_a = party_a if party_a else "[Party A Not Provided]"94    party_b = party_b if party_b else "[Party B Not Provided]"95    context = context if context else "[Context Not Provided]"96 97    prompt = f"""98    You are a helpful legal assistant. Generate a {doc_type} for:99    1) {party_a}100    2) {party_b}101 102    Context/brief of the agreement:103    {context}.104 105    The document should include:106    - Purpose of the {doc_type}107    - Responsibilities and obligations of each party108    - Confidentiality terms109    - Payment terms (use [To Be Determined] if not specified)110    - Term (duration) and termination111    - Governing law: {country}112    - Jurisdiction: [Appropriate region in {country} if not provided]113    - Signature blocks114 115    Use formal language, but keep it relatively clear and readable.116    For any missing information, use placeholders like [To Be Determined].117    Include a disclaimer that this is a draft and not legally binding until reviewed and signed.118    """119    logger.debug(f"Generated prompt:\n{prompt}")120 121    return generate_with_model(prompt, max_length=1400, temperature=0.3)122 123def review_legal_document(doc_text, doc_type, party_a, party_b):124    """125    Reviews document: first with rule-based checks, then wording analysis.126    """127    logger.info("Starting document review (rule-based and wording).")128 129    # --- Rule-Based Review ---130    rule_based_prompt = f"""131You are a legal AI assistant reviewing a document. Provide a review,132structured into the following numbered sections. Be concise and factual. Do NOT133use Markdown. Use plain text labels for each section.134 135Document text:136\"\"\"137{doc_text}138\"\"\"139 140Review Sections:141 1421) Parties and Authority:143    - Confirm the full legal names of all parties.144    - Make sure the people signing can legally commit their organizations.145 1462) Scope of Work / Obligations:147    - Check that the contract clearly describes what each side must do.148    - Look for deadlines, milestones, or deliverables.149    - Ensure everything is realistic and not overly vague.150 1513) Definitions and Key Terms:152    - See if there's a section that explains important terms.153    - Ensure those terms are used the same way throughout the contract.154    - Avoid or clarify any ambiguous language.155 1564) Payment Terms (If Applicable):157    - Check how much is owed, the currency, and when it's due.158    - Look for penalties, interest, or late fees.159    - Note how and when invoices are sent or paid.160 1615) Term and Termination:162    - Identify when the contract starts and ends.163    - Understand how it can be renewed.164    - See the conditions and notice required for ending the contract early.165 1666) Intellectual Property (IP) Rights:167    - Confirm who owns any work created under the agreement.168    - Note if licenses are granted for using the IP, and for how long.169 1707) Confidentiality and Privacy:171    - Check what is considered confidential information.172    - Look for exceptions (like already public info).173    - See how long the confidentiality rules apply.174 1758) Warranties and Representations:176    - Note any performance guarantees or quality promises.177    - Look for disclaimers (like "as is" clauses).178 1799) Indemnification:180    - See who will pay legal costs or damages if there's a lawsuit or claim.181    - Check any limits on what's covered.182 18310) Limitation of Liability:184    - Check if there's a maximum amount one side can claim in damages.185    - Look for excluded damages, like lost profits.186 18711) Dispute Resolution and Governing Law:188    - See if disputes go to arbitration, mediation, or court.189    - Note which state or country's laws will apply.190 19112) Force Majeure (Unforeseen Events):192    - Look for events like natural disasters or war that could suspend obligations.193    - See if there are notice requirements for these events.194 19513) Notices and Amendments:196    - Check how official notices must be sent (email, mail, etc.).197    - Find out how to properly change the contract (in writing, signatures, etc.).198 19914) Entire Agreement and Severability:200    - Confirm that this contract replaces all previous agreements.201    - Ensure that if one clause is invalid, the rest still stands.202 20315) Signatures and Dates:204    - Make sure the right people sign in their proper roles.205    - Verify the date of signature and when the contract goes into effect.206 20716) Ambiguities, Contradictions, and Hidden Clauses:208    - Watch for contradictory statements or clauses that conflict.209    - Beware of vague phrases like "best efforts" without clear guidelines.210    - Check for hidden or "buried" clauses in fine print or attachments.211 21217) Compliance and Regulatory Alignment:213    - Ensure the contract follows relevant laws and rules.214    - Check for industry-specific requirements.215 21618) Practical Considerations:217    - Make sure deadlines and other requirements are doable.218    - Confirm all negotiations are reflected in writing.219    - Avoid blank or undefined items (like fees or dates "to be decided").220"""221    logger.debug(f"Generated rule-based review prompt:\n{rule_based_prompt}")222 223    try:224        rule_based_review = generate_with_model(rule_based_prompt, max_length=2000, temperature=0.3)225    except Exception as e:226        logger.exception("Error during rule-based review.")227        return f"Error during rule-based review: {e}"228 229    # --- Wording Analysis ---230    wording_analysis_prompt = f"""231You are a legal AI assistant. Analyze the following legal document for its wording:232 233Document text:234\"\"\"235{doc_text}236\"\"\"237 238Provide a comprehensive analysis of the document's wording, covering these aspects for the ENTIRE document text:239 2401. **Clarity and Precision:** Identify ambiguous or vague language, and suggest improvements.2412. **Readability:** Assess the overall readability and suggest improvements for clarity, including sentence structure and complexity.2423. **Formal Tone:** Check if the language maintains a formal and professional tone appropriate for a legal document, and suggest changes if needed.2434. **Consistency:** Ensure consistent use of terms and phrasing throughout the document. Point out any inconsistencies.2445. **Redundancy:** Identify any unnecessary repetition of words or phrases.2456. **Jargon and Technical Terms:** Identify jargon or technical terms that might be unclear to a non-expert, and suggest clearer alternatives where appropriate.2467. **Overall Recommendations:** Give overall recommendations for improving the document's wording.247 248Provide your analysis in plain text, without using Markdown. Label each section of your analysis clearly (e.g., "Clarity and Precision:", "Readability:", etc.).249"""250    logger.debug(f"Generated wording analysis prompt:\n{wording_analysis_prompt}")251 252    try:253        wording_analysis = generate_with_model(wording_analysis_prompt, max_length=1000, temperature=0.3)254    except Exception as e:255        logger.exception("Error during wording analysis.")256        return f"Error during wording analysis: {e}"257 258    combined_review = f"Rule-Based Analysis:\n\n{rule_based_review}\n\nWording Analysis:\n\n{wording_analysis}"259    return combined_review260 261###############################################################################262# 4) File Parsing (PDF, DOCX)263###############################################################################264 265def parse_bytesio(file_data: BytesIO) -> str:266    """Parses a BytesIO object representing a PDF or DOCX."""267    logger.info("Parsing BytesIO object...")268    try:269        # Attempt to determine file type from content270        try:271            doc_obj = docx.Document(file_data)272            return "\n".join([para.text for para in doc_obj.paragraphs]).strip()273        except docx.opc.exceptions.PackageNotFoundError:274            logger.info("BytesIO is not DOCX, trying PDF.")275            file_data.seek(0)276            try:277                pdf_reader = PyPDF2.PdfReader(file_data)278                return "\n".join([page.extract_text() for page in pdf_reader.pages if page.extract_text()]).strip()279            except Exception as e:280                logger.exception(f"Error parsing BytesIO as PDF: {e}")281                return f"Error parsing BytesIO as PDF: {e}"282        except Exception as e:283            logger.exception(f"Error processing BytesIO: {e}")284            return f"Error processing file content: {e}"285    except Exception as e:286        logger.exception(f"Error parsing BytesIO: {e}")287        return f"Error parsing BytesIO: {e}"288 289def parse_uploaded_file_path(file_data) -> str:290    """Takes file data, determines type, extracts text."""291    if not file_data:292        logger.warning("No file provided.")293        return ""294    if isinstance(file_data, str):295        file_path = file_data296        logger.info(f"Received filepath: {file_path}")297    elif isinstance(file_data, dict) and 'name' in file_data:298        file_path = file_data['name']299        logger.info(f"Received file object with name: {file_path}")300    elif isinstance(file_data, (BytesIO, IOBase)):301        return parse_bytesio(file_data)302    else:303        logger.error(f"Unexpected file_data type: {type(file_data)}")304        return "Error: Unexpected file data format."305 306    logger.info(f"Attempting to parse file at {file_path}")307    try:308        _, ext = os.path.splitext(file_path)309        ext = ext.lower()310        if ext == ".pdf":311            with open(file_path, "rb") as f:312                pdf_reader = PyPDF2.PdfReader(f)313                return "\n".join([page.extract_text() for page in pdf_reader.pages if page.extract_text()]).strip()314        elif ext == ".docx":315            doc_obj = docx.Document(file_path)316            return "\n".join([para.text for para in doc_obj.paragraphs]).strip()317        else:318            return "Unsupported file format."319    except Exception as e:320        logger.exception(f"Error parsing file: {e}")321        return f"Error parsing file: {e}"322    finally:323        pass324 325###############################################################################326# 5) DOCX Creation and Saving327###############################################################################328 329def clean_markdown(text):330    """Removes common Markdown formatting."""331    if not text: return ""332    text = re.sub(r'^#+\s+', '', text, flags=re.MULTILINE)333    text = re.sub(r'(\*\*|__)(.*?)(\*\*|__)', r'\2', text)334    text = re.sub(r'(\*|_)(.*?)(\*|_)', r'\2', text)335    text = re.sub(r'^[\-\+\*]\s+', '', text, flags=re.MULTILINE)336    text = re.sub(r'^\d+\.\s+', '', text, flags=re.MULTILINE)337    text = re.sub(r'^[-_*]{3,}$', '', text, flags=re.MULTILINE)338    text = re.sub(r'!\[(.*?)\]\((.*?)\)', '', text)339    text = re.sub(r'\[(.*?)\]\((.*?)\)', r'\1', text)340    return text.strip()341 342def create_and_save_docx(doc_text, review_text=None, doc_type="Unknown", party_a="Party A", party_b="Party B"):343    """Creates DOCX, adds review, saves to temp file, returns path."""344    logger.debug("Creating and saving DOCX.")345    document = docx.Document()346 347    now = datetime.datetime.now()348    timestamp = now.strftime("%Y%m%d_%H%M%S")349    file_name = f"HF_AI_Review_{doc_type}_{timestamp}.docx"350 351    title = f"DocumentCogito Analysis of {doc_type} between companies {party_a} and {party_b}"352    document.add_heading(title, level=1)353 354    if doc_text:355        document.add_heading("Generated Document", level=2)356        for para in clean_markdown(doc_text).split("\n"):357            document.add_paragraph(para)358 359    if review_text:360        document.add_heading("LLM Review", level=2)361        for section in review_text.split("\n\n"):  # Split into sections362            if section.startswith("Rule-Based Analysis:"):363                analysis_heading = document.add_paragraph()364                analysis_run = analysis_heading.add_run("Rule-Based Analysis")365                analysis_run.font.size = Pt(14)366                analysis_run.font.color.rgb = RGBColor(0xFF, 0x00, 0x00)367                for para in section[len("Rule-Based Analysis:"):].split("\n"):368                    if re.match(r"^\d+\)", para):  # Check for numbered points369                        p = document.add_paragraph(style='List Number')370                        p.add_run(para).font.color.rgb = RGBColor(0xFF, 0x00, 0x00) #red371                    else:372                        document.add_paragraph(para)373 374            elif section.startswith("Wording Analysis:"):375                analysis_heading = document.add_paragraph()376                analysis_run = analysis_heading.add_run("Wording Analysis")377                analysis_run.font.size = Pt(14)378                analysis_run.font.color.rgb = RGBColor(0xFF, 0x00, 0x00)379                for para in section[len("Wording Analysis:"):].split("\n"):380                    document.add_paragraph(para) # black, no numbering381            else:  # Other sections (if any)382                document.add_paragraph(section)383 384    with tempfile.NamedTemporaryFile(delete=False, suffix=f"_{file_name}") as tmpfile:385        document.save(tmpfile.name)386        logger.debug(f"DOCX saved to: {tmpfile.name}")387        return tmpfile.name388 389###############################################################################390# 6) Gradio Interface Functions391###############################################################################392 393def generate_document_interface(doc_type, party_a, party_b, context, country):394    """Handles document generation."""395    logger.info(f"User requested doc generation: {doc_type}, {country}")396    doc_text = generate_legal_document(doc_type, party_a, party_b, context, country)397    if doc_text.startswith("Error"):398      return doc_text, None399    docx_file_path = create_and_save_docx(doc_text, doc_type=doc_type, party_a=party_a, party_b=party_b)400    return doc_text, docx_file_path401 402def review_document_interface(file_data, doc_type, party_a, party_b):403    """Handles document review."""404    logger.info("User requested review.")405    if not file_data:406        return "No file uploaded.", None407 408    original_text = parse_uploaded_file_path(file_data)409    if original_text.startswith("Error") or original_text.startswith("Unsupported"):410        return original_text, None411 412    review_text = review_legal_document(original_text, doc_type, party_a, party_b)413    if review_text.startswith("Error"):414        return review_text, None415 416    docx_file_path = create_and_save_docx(None, review_text, doc_type, party_a, party_b)417    return review_text, docx_file_path418 419###############################################################################420# 7) Build & Launch Gradio App421###############################################################################422# Define custom CSS in a string.423custom_css = """424.tab-one {425    background-color: #D1EEFC; /* Light blue */426    color: #333;427}428.tab-two {429    background-color: #FCEED1; /* Light orange */430    color: #333;431}432/* If you want to style the tab label differently, you may need to target433   specific child elements (like a .tab__header) within the class. */434"""435 436def build_app():437    with gr.Blocks(css=custom_css) as demo:438        gr.Markdown(439            """440            # UST Global Legal Document Analyzer (Hugging Face Version)441 442            **Review an Existing MOU, SOW, MSA in PDF/DOCX format**: Upload a document for analysis.443 444            **Disclaimer**: This tool provides assistance but is not a substitute for professional legal advice.445            """446        )447        with gr.Tabs(selected=1):448          with gr.Tab("Generate Document", visible=False):449              doc_type = gr.Dropdown(label="Document Type", choices=["MOU", "MSA", "SoW", "NDA"], value="MOU")450              party_a = gr.Textbox(label="Party A Name", placeholder="e.g., Tech Innovations LLC")451              party_b = gr.Textbox(label="Party B Name", placeholder="e.g., Global Consulting Corp")452              context = gr.Textbox(label="Context/Brief", placeholder="Short summary of the agreement...")453              country = gr.Dropdown(label="Governing Law (Country)", choices=["India", "Malaysia", "US", "UK", "Singapore", "Japan"], value="India")454              gen_button = gr.Button("Generate Document")455              gen_output_text = gr.Textbox(label="Generated Document", lines=15, placeholder="Generated document will appear here...")456              gen_output_file = gr.File(label="Download DOCX", type="filepath")457              gen_button.click(458                  generate_document_interface,459                  inputs=[doc_type, party_a, party_b, context, country],460                  outputs=[gen_output_text, gen_output_file]461              )462 463          with gr.Tab("Review Document", elem_classes="tab-one", id=1):464              # Hidden inputs to store values from Generate tab465              doc_type_review = gr.Dropdown(label="Document Type", choices=["MOU", "MSA", "SoW", "NDA"], value="MOU", visible=False)466              party_a_review = gr.Textbox(label="Party A Name", visible=False)467              party_b_review = gr.Textbox(label="Party B Name", visible=False)468 469              file_input = gr.File(label="Upload PDF/DOCX for Review", type="filepath")470              review_button = gr.Button("Review Document")471              review_output_text = gr.Textbox(label="Review", lines=15, placeholder="Review will appear here...")472              review_output_file = gr.File(label="Download Reviewed DOCX", type="filepath")473              review_button.click(474                  review_document_interface,475                  inputs=[file_input, doc_type_review, party_a_review, party_b_review],476                  outputs=[review_output_text, review_output_file]477              )478              # Copy values from Generate to Review tab (hidden fields)479              gen_button.click(lambda x, y, z: (x, y, z), [doc_type, party_a, party_b], [doc_type_review, party_a_review, party_b_review])480 481          gr.Markdown("**Note:** Scanned PDFs may not parse correctly. .docx is generally preferred.")482    return demo483 484# For Hugging Face Spaces deployment485if __name__ == "__main__":486#    create_requirements_file()487    logger.info("Initializing Gradio interface...")488    demo = build_app()489    logger.info("Launching Gradio app.")490    demo.launch(debug=True,share=False)