Coder19/interview_system
0
1import os2import fitz # PyMuPDF3from typing import Optional, Tuple4from transformers import pipeline, Pipeline5from fastapi import FastAPI, UploadFile, File, HTTPException6from fastapi.responses import JSONResponse7import io8import tempfile9import logging10from dotenv import load_dotenv11 12# Configure logging13logging.basicConfig(level=logging.INFO)14logger = logging.getLogger(__name__)15 16# Load environment variables17load_dotenv()18HF_TOKEN = os.getenv("HF_TOKEN")19 20# Initialize summarizer21summarizer: Optional[Pipeline] = None22 23def initialize_summarizer() -> None:24 """Initialize the summarization model with error handling."""25 global summarizer26 try:27 logger.info("Loading public model: sshleifer/distilbart-cnn-12-6")28 summarizer = pipeline("summarization", model="sshleifer/distilbart-cnn-12-6")29 logger.info("Model loaded successfully")30 except Exception as e:31 logger.error(f"Failed to load model: {str(e)}")32 raise33 34# Initialize on import35try:36 initialize_summarizer()37except Exception as e:38 logger.warning(f"Could not initialize summarizer: {str(e)}")39 40def extract_text_from_pdf(pdf_bytes: io.BytesIO) -> dict:41 """Extract text from PDF with error handling.42 43 Args:44 pdf_bytes: BytesIO object containing PDF data45 46 Returns:47 Dictionary with 'text' and 'error' keys.48 If successful, 'error' will be None.49 """50 if not pdf_bytes or pdf_bytes.getbuffer().nbytes == 0:51 return {"text": None, "error": "Empty PDF data provided"}52 53 temp_file = None54 try:55 with tempfile.NamedTemporaryFile(delete=False, suffix='.pdf') as temp_file:56 temp_file.write(pdf_bytes.getvalue())57 temp_file_path = temp_file.name58 59 with fitz.open(temp_file_path) as doc:60 text = "".join(page.get_text() for page in doc)61 62 if not text.strip():63 return {"text": None, "error": "No text could be extracted from the PDF"}64 65 return {"text": text, "error": None}66 67 except Exception as e:68 error_msg = str(e)69 logger.error(f"Error extracting text from PDF: {error_msg}")70 return {"text": None, "error": f"Error processing PDF: {error_msg}"}71 72 finally:73 if temp_file and 'temp_file_path' in locals() and os.path.exists(temp_file_path):74 try:75 os.unlink(temp_file_path)76 except Exception as e:77 logger.warning(f"Could not delete temporary file: {str(e)}")78 79def summarize_text(text: str, max_chunk_len: int = 1000) -> dict:80 """Summarize text in chunks with error handling.81 82 Args:83 text: The text to summarize84 max_chunk_len: Maximum length of each chunk85 86 Returns:87 Dictionary with 'summary' and 'error' keys.88 If successful, 'error' will be None.89 """90 if not text or not isinstance(text, str):91 return {"summary": None, "error": "Invalid text input"}92 93 text = text.strip()94 if not text:95 return {"summary": None, "error": "Empty text provided"}96 97 if not summarizer:98 return {"summary": None, "error": "Summarizer not initialized"}99 100 try:101 summaries = []102 for i in range(0, len(text), max_chunk_len):103 chunk = text[i:i + max_chunk_len]104 try:105 result = summarizer(chunk, max_length=130, min_length=30, do_sample=False)106 if result and isinstance(result, list) and len(result) > 0:107 summary = result[0].get('summary_text', '')108 if summary:109 summaries.append(summary)110 except Exception as chunk_error:111 logger.warning(f"Error summarizing chunk {i//max_chunk_len + 1}: {str(chunk_error)}")112 continue113 114 if not summaries:115 return {"summary": None, "error": "Failed to generate any summaries"}116 117 return {"summary": " ".join(summaries), "error": None}118 119 except Exception as e:120 error_msg = str(e)121 logger.error(f"Error in summarization: {error_msg}")122 return {"summary": None, "error": error_msg}123 