AITestingWorkSpace/DocumentAnalysis
0
1import logging2import os3import docx4import PyPDF25from docx.shared import RGBColor, Pt6from io import BytesIO, IOBase7import tempfile8import re9import datetime10import torch11 12import gradio as gr13from transformers import AutoModelForCausalLM, AutoTokenizer14import huggingface_hub15 16###############################################################################17# 1) Logging Configuration18###############################################################################19logging.basicConfig(20 level=logging.INFO, # or logging.DEBUG for more verbose logs21 format="%(asctime)s [%(levelname)s] %(name)s - %(message)s"22)23logger = logging.getLogger("LLM-Legal-App")24 25###############################################################################26# 2) Initialize Hugging Face Model27###############################################################################28def initialize_model():29 """Initialize the phi-2 model and tokenizer from HuggingFace."""30 logger.info("Initializing phi-2 model and tokenizer...")31 try:32 # Access token might be needed for some models33 # token = huggingface_hub.get_token()34 35 model_name = "microsoft/phi-2"36 tokenizer = AutoTokenizer.from_pretrained(model_name)37 model = AutoModelForCausalLM.from_pretrained(38 model_name,39 torch_dtype=torch.float16,40 device_map="auto",41 trust_remote_code=True42 )43 logger.info("Successfully initialized phi-2 model and tokenizer.")44 return model, tokenizer45 except Exception as e:46 logger.exception("Error initializing Hugging Face model.")47 raise ValueError(f"Failed to initialize model: {e}")48 49# Initialize model and tokenizer50model, tokenizer = initialize_model()51 52###############################################################################53# 3) LLM Utility Functions (Generation & Review)54###############################################################################55def generate_with_model(prompt, max_length=1400, temperature=0.3):56 """Generate text using the Hugging Face model."""57 logger.info("Generating text with phi-2 model.")58 59 try:60 inputs = tokenizer(prompt, return_tensors="pt").to(model.device)61 62 # Generate with parameters similar to the original OpenAI call63 generation_config = {64 "max_new_tokens": max_length,65 "temperature": temperature,66 "top_p": 0.9,67 "do_sample": temperature > 0,68 "pad_token_id": tokenizer.eos_token_id69 }70 71 with torch.no_grad():72 outputs = model.generate(**inputs, **generation_config)73 74 response = tokenizer.decode(outputs[0], skip_special_tokens=True)75 76 # Remove the prompt from the response77 if response.startswith(prompt):78 response = response[len(prompt):].strip()79 80 logger.info("Text generation complete.")81 return response82 83 except Exception as e:84 logger.exception("Error during text generation.")85 return f"Error generating text: {e}"86 87def generate_legal_document(doc_type, party_a, party_b, context, country):88 """89 Uses DocumentCogito to generate a legal document. Returns the document text.90 """91 logger.info(f"Starting generation for doc_type={doc_type!r}.")92 # Fill placeholders if fields are missing93 party_a = party_a if party_a else "[Party A Not Provided]"94 party_b = party_b if party_b else "[Party B Not Provided]"95 context = context if context else "[Context Not Provided]"96 97 prompt = f"""98 You are a helpful legal assistant. Generate a {doc_type} for:99 1) {party_a}100 2) {party_b}101 102 Context/brief of the agreement:103 {context}.104 105 The document should include:106 - Purpose of the {doc_type}107 - Responsibilities and obligations of each party108 - Confidentiality terms109 - Payment terms (use [To Be Determined] if not specified)110 - Term (duration) and termination111 - Governing law: {country}112 - Jurisdiction: [Appropriate region in {country} if not provided]113 - Signature blocks114 115 Use formal language, but keep it relatively clear and readable.116 For any missing information, use placeholders like [To Be Determined].117 Include a disclaimer that this is a draft and not legally binding until reviewed and signed.118 """119 logger.debug(f"Generated prompt:\n{prompt}")120 121 return generate_with_model(prompt, max_length=1400, temperature=0.3)122 123def review_legal_document(doc_text, doc_type, party_a, party_b):124 """125 Reviews document: first with rule-based checks, then wording analysis.126 """127 logger.info("Starting document review (rule-based and wording).")128 129 # --- Rule-Based Review ---130 rule_based_prompt = f"""131You are a legal AI assistant reviewing a document. Provide a review,132structured into the following numbered sections. Be concise and factual. Do NOT133use Markdown. Use plain text labels for each section.134 135Document text:136\"\"\"137{doc_text}138\"\"\"139 140Review Sections:141 1421) Parties and Authority:143 - Confirm the full legal names of all parties.144 - Make sure the people signing can legally commit their organizations.145 1462) Scope of Work / Obligations:147 - Check that the contract clearly describes what each side must do.148 - Look for deadlines, milestones, or deliverables.149 - Ensure everything is realistic and not overly vague.150 1513) Definitions and Key Terms:152 - See if there's a section that explains important terms.153 - Ensure those terms are used the same way throughout the contract.154 - Avoid or clarify any ambiguous language.155 1564) Payment Terms (If Applicable):157 - Check how much is owed, the currency, and when it's due.158 - Look for penalties, interest, or late fees.159 - Note how and when invoices are sent or paid.160 1615) Term and Termination:162 - Identify when the contract starts and ends.163 - Understand how it can be renewed.164 - See the conditions and notice required for ending the contract early.165 1666) Intellectual Property (IP) Rights:167 - Confirm who owns any work created under the agreement.168 - Note if licenses are granted for using the IP, and for how long.169 1707) Confidentiality and Privacy:171 - Check what is considered confidential information.172 - Look for exceptions (like already public info).173 - See how long the confidentiality rules apply.174 1758) Warranties and Representations:176 - Note any performance guarantees or quality promises.177 - Look for disclaimers (like "as is" clauses).178 1799) Indemnification:180 - See who will pay legal costs or damages if there's a lawsuit or claim.181 - Check any limits on what's covered.182 18310) Limitation of Liability:184 - Check if there's a maximum amount one side can claim in damages.185 - Look for excluded damages, like lost profits.186 18711) Dispute Resolution and Governing Law:188 - See if disputes go to arbitration, mediation, or court.189 - Note which state or country's laws will apply.190 19112) Force Majeure (Unforeseen Events):192 - Look for events like natural disasters or war that could suspend obligations.193 - See if there are notice requirements for these events.194 19513) Notices and Amendments:196 - Check how official notices must be sent (email, mail, etc.).197 - Find out how to properly change the contract (in writing, signatures, etc.).198 19914) Entire Agreement and Severability:200 - Confirm that this contract replaces all previous agreements.201 - Ensure that if one clause is invalid, the rest still stands.202 20315) Signatures and Dates:204 - Make sure the right people sign in their proper roles.205 - Verify the date of signature and when the contract goes into effect.206 20716) Ambiguities, Contradictions, and Hidden Clauses:208 - Watch for contradictory statements or clauses that conflict.209 - Beware of vague phrases like "best efforts" without clear guidelines.210 - Check for hidden or "buried" clauses in fine print or attachments.211 21217) Compliance and Regulatory Alignment:213 - Ensure the contract follows relevant laws and rules.214 - Check for industry-specific requirements.215 21618) Practical Considerations:217 - Make sure deadlines and other requirements are doable.218 - Confirm all negotiations are reflected in writing.219 - Avoid blank or undefined items (like fees or dates "to be decided").220"""221 logger.debug(f"Generated rule-based review prompt:\n{rule_based_prompt}")222 223 try:224 rule_based_review = generate_with_model(rule_based_prompt, max_length=2000, temperature=0.3)225 except Exception as e:226 logger.exception("Error during rule-based review.")227 return f"Error during rule-based review: {e}"228 229 # --- Wording Analysis ---230 wording_analysis_prompt = f"""231You are a legal AI assistant. Analyze the following legal document for its wording:232 233Document text:234\"\"\"235{doc_text}236\"\"\"237 238Provide a comprehensive analysis of the document's wording, covering these aspects for the ENTIRE document text:239 2401. **Clarity and Precision:** Identify ambiguous or vague language, and suggest improvements.2412. **Readability:** Assess the overall readability and suggest improvements for clarity, including sentence structure and complexity.2423. **Formal Tone:** Check if the language maintains a formal and professional tone appropriate for a legal document, and suggest changes if needed.2434. **Consistency:** Ensure consistent use of terms and phrasing throughout the document. Point out any inconsistencies.2445. **Redundancy:** Identify any unnecessary repetition of words or phrases.2456. **Jargon and Technical Terms:** Identify jargon or technical terms that might be unclear to a non-expert, and suggest clearer alternatives where appropriate.2467. **Overall Recommendations:** Give overall recommendations for improving the document's wording.247 248Provide your analysis in plain text, without using Markdown. Label each section of your analysis clearly (e.g., "Clarity and Precision:", "Readability:", etc.).249"""250 logger.debug(f"Generated wording analysis prompt:\n{wording_analysis_prompt}")251 252 try:253 wording_analysis = generate_with_model(wording_analysis_prompt, max_length=1000, temperature=0.3)254 except Exception as e:255 logger.exception("Error during wording analysis.")256 return f"Error during wording analysis: {e}"257 258 combined_review = f"Rule-Based Analysis:\n\n{rule_based_review}\n\nWording Analysis:\n\n{wording_analysis}"259 return combined_review260 261###############################################################################262# 4) File Parsing (PDF, DOCX)263###############################################################################264 265def parse_bytesio(file_data: BytesIO) -> str:266 """Parses a BytesIO object representing a PDF or DOCX."""267 logger.info("Parsing BytesIO object...")268 try:269 # Attempt to determine file type from content270 try:271 doc_obj = docx.Document(file_data)272 return "\n".join([para.text for para in doc_obj.paragraphs]).strip()273 except docx.opc.exceptions.PackageNotFoundError:274 logger.info("BytesIO is not DOCX, trying PDF.")275 file_data.seek(0)276 try:277 pdf_reader = PyPDF2.PdfReader(file_data)278 return "\n".join([page.extract_text() for page in pdf_reader.pages if page.extract_text()]).strip()279 except Exception as e:280 logger.exception(f"Error parsing BytesIO as PDF: {e}")281 return f"Error parsing BytesIO as PDF: {e}"282 except Exception as e:283 logger.exception(f"Error processing BytesIO: {e}")284 return f"Error processing file content: {e}"285 except Exception as e:286 logger.exception(f"Error parsing BytesIO: {e}")287 return f"Error parsing BytesIO: {e}"288 289def parse_uploaded_file_path(file_data) -> str:290 """Takes file data, determines type, extracts text."""291 if not file_data:292 logger.warning("No file provided.")293 return ""294 if isinstance(file_data, str):295 file_path = file_data296 logger.info(f"Received filepath: {file_path}")297 elif isinstance(file_data, dict) and 'name' in file_data:298 file_path = file_data['name']299 logger.info(f"Received file object with name: {file_path}")300 elif isinstance(file_data, (BytesIO, IOBase)):301 return parse_bytesio(file_data)302 else:303 logger.error(f"Unexpected file_data type: {type(file_data)}")304 return "Error: Unexpected file data format."305 306 logger.info(f"Attempting to parse file at {file_path}")307 try:308 _, ext = os.path.splitext(file_path)309 ext = ext.lower()310 if ext == ".pdf":311 with open(file_path, "rb") as f:312 pdf_reader = PyPDF2.PdfReader(f)313 return "\n".join([page.extract_text() for page in pdf_reader.pages if page.extract_text()]).strip()314 elif ext == ".docx":315 doc_obj = docx.Document(file_path)316 return "\n".join([para.text for para in doc_obj.paragraphs]).strip()317 else:318 return "Unsupported file format."319 except Exception as e:320 logger.exception(f"Error parsing file: {e}")321 return f"Error parsing file: {e}"322 finally:323 pass324 325###############################################################################326# 5) DOCX Creation and Saving327###############################################################################328 329def clean_markdown(text):330 """Removes common Markdown formatting."""331 if not text: return ""332 text = re.sub(r'^#+\s+', '', text, flags=re.MULTILINE)333 text = re.sub(r'(\*\*|__)(.*?)(\*\*|__)', r'\2', text)334 text = re.sub(r'(\*|_)(.*?)(\*|_)', r'\2', text)335 text = re.sub(r'^[\-\+\*]\s+', '', text, flags=re.MULTILINE)336 text = re.sub(r'^\d+\.\s+', '', text, flags=re.MULTILINE)337 text = re.sub(r'^[-_*]{3,}$', '', text, flags=re.MULTILINE)338 text = re.sub(r'!\[(.*?)\]\((.*?)\)', '', text)339 text = re.sub(r'\[(.*?)\]\((.*?)\)', r'\1', text)340 return text.strip()341 342def create_and_save_docx(doc_text, review_text=None, doc_type="Unknown", party_a="Party A", party_b="Party B"):343 """Creates DOCX, adds review, saves to temp file, returns path."""344 logger.debug("Creating and saving DOCX.")345 document = docx.Document()346 347 now = datetime.datetime.now()348 timestamp = now.strftime("%Y%m%d_%H%M%S")349 file_name = f"HF_AI_Review_{doc_type}_{timestamp}.docx"350 351 title = f"DocumentCogito Analysis of {doc_type} between companies {party_a} and {party_b}"352 document.add_heading(title, level=1)353 354 if doc_text:355 document.add_heading("Generated Document", level=2)356 for para in clean_markdown(doc_text).split("\n"):357 document.add_paragraph(para)358 359 if review_text:360 document.add_heading("LLM Review", level=2)361 for section in review_text.split("\n\n"): # Split into sections362 if section.startswith("Rule-Based Analysis:"):363 analysis_heading = document.add_paragraph()364 analysis_run = analysis_heading.add_run("Rule-Based Analysis")365 analysis_run.font.size = Pt(14)366 analysis_run.font.color.rgb = RGBColor(0xFF, 0x00, 0x00)367 for para in section[len("Rule-Based Analysis:"):].split("\n"):368 if re.match(r"^\d+\)", para): # Check for numbered points369 p = document.add_paragraph(style='List Number')370 p.add_run(para).font.color.rgb = RGBColor(0xFF, 0x00, 0x00) #red371 else:372 document.add_paragraph(para)373 374 elif section.startswith("Wording Analysis:"):375 analysis_heading = document.add_paragraph()376 analysis_run = analysis_heading.add_run("Wording Analysis")377 analysis_run.font.size = Pt(14)378 analysis_run.font.color.rgb = RGBColor(0xFF, 0x00, 0x00)379 for para in section[len("Wording Analysis:"):].split("\n"):380 document.add_paragraph(para) # black, no numbering381 else: # Other sections (if any)382 document.add_paragraph(section)383 384 with tempfile.NamedTemporaryFile(delete=False, suffix=f"_{file_name}") as tmpfile:385 document.save(tmpfile.name)386 logger.debug(f"DOCX saved to: {tmpfile.name}")387 return tmpfile.name388 389###############################################################################390# 6) Gradio Interface Functions391###############################################################################392 393def generate_document_interface(doc_type, party_a, party_b, context, country):394 """Handles document generation."""395 logger.info(f"User requested doc generation: {doc_type}, {country}")396 doc_text = generate_legal_document(doc_type, party_a, party_b, context, country)397 if doc_text.startswith("Error"):398 return doc_text, None399 docx_file_path = create_and_save_docx(doc_text, doc_type=doc_type, party_a=party_a, party_b=party_b)400 return doc_text, docx_file_path401 402def review_document_interface(file_data, doc_type, party_a, party_b):403 """Handles document review."""404 logger.info("User requested review.")405 if not file_data:406 return "No file uploaded.", None407 408 original_text = parse_uploaded_file_path(file_data)409 if original_text.startswith("Error") or original_text.startswith("Unsupported"):410 return original_text, None411 412 review_text = review_legal_document(original_text, doc_type, party_a, party_b)413 if review_text.startswith("Error"):414 return review_text, None415 416 docx_file_path = create_and_save_docx(None, review_text, doc_type, party_a, party_b)417 return review_text, docx_file_path418 419###############################################################################420# 7) Build & Launch Gradio App421###############################################################################422# Define custom CSS in a string.423custom_css = """424.tab-one {425 background-color: #D1EEFC; /* Light blue */426 color: #333;427}428.tab-two {429 background-color: #FCEED1; /* Light orange */430 color: #333;431}432/* If you want to style the tab label differently, you may need to target433 specific child elements (like a .tab__header) within the class. */434"""435 436def build_app():437 with gr.Blocks(css=custom_css) as demo:438 gr.Markdown(439 """440 # UST Global Legal Document Analyzer (Hugging Face Version)441 442 **Review an Existing MOU, SOW, MSA in PDF/DOCX format**: Upload a document for analysis.443 444 **Disclaimer**: This tool provides assistance but is not a substitute for professional legal advice.445 """446 )447 with gr.Tabs(selected=1):448 with gr.Tab("Generate Document", visible=False):449 doc_type = gr.Dropdown(label="Document Type", choices=["MOU", "MSA", "SoW", "NDA"], value="MOU")450 party_a = gr.Textbox(label="Party A Name", placeholder="e.g., Tech Innovations LLC")451 party_b = gr.Textbox(label="Party B Name", placeholder="e.g., Global Consulting Corp")452 context = gr.Textbox(label="Context/Brief", placeholder="Short summary of the agreement...")453 country = gr.Dropdown(label="Governing Law (Country)", choices=["India", "Malaysia", "US", "UK", "Singapore", "Japan"], value="India")454 gen_button = gr.Button("Generate Document")455 gen_output_text = gr.Textbox(label="Generated Document", lines=15, placeholder="Generated document will appear here...")456 gen_output_file = gr.File(label="Download DOCX", type="filepath")457 gen_button.click(458 generate_document_interface,459 inputs=[doc_type, party_a, party_b, context, country],460 outputs=[gen_output_text, gen_output_file]461 )462 463 with gr.Tab("Review Document", elem_classes="tab-one", id=1):464 # Hidden inputs to store values from Generate tab465 doc_type_review = gr.Dropdown(label="Document Type", choices=["MOU", "MSA", "SoW", "NDA"], value="MOU", visible=False)466 party_a_review = gr.Textbox(label="Party A Name", visible=False)467 party_b_review = gr.Textbox(label="Party B Name", visible=False)468 469 file_input = gr.File(label="Upload PDF/DOCX for Review", type="filepath")470 review_button = gr.Button("Review Document")471 review_output_text = gr.Textbox(label="Review", lines=15, placeholder="Review will appear here...")472 review_output_file = gr.File(label="Download Reviewed DOCX", type="filepath")473 review_button.click(474 review_document_interface,475 inputs=[file_input, doc_type_review, party_a_review, party_b_review],476 outputs=[review_output_text, review_output_file]477 )478 # Copy values from Generate to Review tab (hidden fields)479 gen_button.click(lambda x, y, z: (x, y, z), [doc_type, party_a, party_b], [doc_type_review, party_a_review, party_b_review])480 481 gr.Markdown("**Note:** Scanned PDFs may not parse correctly. .docx is generally preferred.")482 return demo483 484# For Hugging Face Spaces deployment485if __name__ == "__main__":486# create_requirements_file()487 logger.info("Initializing Gradio interface...")488 demo = build_app()489 logger.info("Launching Gradio app.")490 demo.launch(debug=True,share=False)