Team Ai
Apppublic

Coder19/interview_system

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
summarization.py123 linesDownload Raw Back to interview_prep
1import os2import fitz  # PyMuPDF3from typing import Optional, Tuple4from transformers import pipeline, Pipeline5from fastapi import FastAPI, UploadFile, File, HTTPException6from fastapi.responses import JSONResponse7import io8import tempfile9import logging10from dotenv import load_dotenv11 12# Configure logging13logging.basicConfig(level=logging.INFO)14logger = logging.getLogger(__name__)15 16# Load environment variables17load_dotenv()18HF_TOKEN = os.getenv("HF_TOKEN")19 20# Initialize summarizer21summarizer: Optional[Pipeline] = None22 23def initialize_summarizer() -> None:24    """Initialize the summarization model with error handling."""25    global summarizer26    try:27        logger.info("Loading public model: sshleifer/distilbart-cnn-12-6")28        summarizer = pipeline("summarization", model="sshleifer/distilbart-cnn-12-6")29        logger.info("Model loaded successfully")30    except Exception as e:31        logger.error(f"Failed to load model: {str(e)}")32        raise33 34# Initialize on import35try:36    initialize_summarizer()37except Exception as e:38    logger.warning(f"Could not initialize summarizer: {str(e)}")39 40def extract_text_from_pdf(pdf_bytes: io.BytesIO) -> dict:41    """Extract text from PDF with error handling.42    43    Args:44        pdf_bytes: BytesIO object containing PDF data45        46    Returns:47        Dictionary with 'text' and 'error' keys.48        If successful, 'error' will be None.49    """50    if not pdf_bytes or pdf_bytes.getbuffer().nbytes == 0:51        return {"text": None, "error": "Empty PDF data provided"}52        53    temp_file = None54    try:55        with tempfile.NamedTemporaryFile(delete=False, suffix='.pdf') as temp_file:56            temp_file.write(pdf_bytes.getvalue())57            temp_file_path = temp_file.name58            59        with fitz.open(temp_file_path) as doc:60            text = "".join(page.get_text() for page in doc)61            62        if not text.strip():63            return {"text": None, "error": "No text could be extracted from the PDF"}64            65        return {"text": text, "error": None}66        67    except Exception as e:68        error_msg = str(e)69        logger.error(f"Error extracting text from PDF: {error_msg}")70        return {"text": None, "error": f"Error processing PDF: {error_msg}"}71        72    finally:73        if temp_file and 'temp_file_path' in locals() and os.path.exists(temp_file_path):74            try:75                os.unlink(temp_file_path)76            except Exception as e:77                logger.warning(f"Could not delete temporary file: {str(e)}")78 79def summarize_text(text: str, max_chunk_len: int = 1000) -> dict:80    """Summarize text in chunks with error handling.81    82    Args:83        text: The text to summarize84        max_chunk_len: Maximum length of each chunk85        86    Returns:87        Dictionary with 'summary' and 'error' keys.88        If successful, 'error' will be None.89    """90    if not text or not isinstance(text, str):91        return {"summary": None, "error": "Invalid text input"}92        93    text = text.strip()94    if not text:95        return {"summary": None, "error": "Empty text provided"}96        97    if not summarizer:98        return {"summary": None, "error": "Summarizer not initialized"}99        100    try:101        summaries = []102        for i in range(0, len(text), max_chunk_len):103            chunk = text[i:i + max_chunk_len]104            try:105                result = summarizer(chunk, max_length=130, min_length=30, do_sample=False)106                if result and isinstance(result, list) and len(result) > 0:107                    summary = result[0].get('summary_text', '')108                    if summary:109                        summaries.append(summary)110            except Exception as chunk_error:111                logger.warning(f"Error summarizing chunk {i//max_chunk_len + 1}: {str(chunk_error)}")112                continue113                114        if not summaries:115            return {"summary": None, "error": "Failed to generate any summaries"}116            117        return {"summary": " ".join(summaries), "error": None}118        119    except Exception as e:120        error_msg = str(e)121        logger.error(f"Error in summarization: {error_msg}")122        return {"summary": None, "error": error_msg}123