bluewhale2025/parseai-document-processor
0
1import nltk2from typing import Dict, List3import json4from datetime import datetime5import heapq6 7class DocumentSummarizer:8 def __init__(self):9 # Set NLTK data path10 nltk_data_paths = [11 '/usr/local/share/nltk_data',12 '/usr/share/nltk_data',13 '/usr/local/nltk_data',14 '/usr/local/lib/nltk_data',15 '/usr/lib/nltk_data',16 '/root/nltk_data',17 '/home/user/nltk_data',18 '/app/nltk_data'19 ]20 21 # Add all possible NLTK data paths22 nltk.data.path = list(dict.fromkeys(nltk_data_paths + nltk.data.path))23 24 # Download NLTK data if not found25 try:26 nltk.download('punkt')27 nltk.download('stopwords')28 nltk.download('wordnet')29 nltk.download('averaged_perceptron_tagger')30 except Exception as e:31 print(f"Warning: NLTK data download failed: {str(e)}")32 33 # 텍스트 분할 크기 설정34 self.chunk_size = 1000 # 토큰 기준35 try:36 self.tokenizer = nltk.data.load('tokenizers/punkt/english.pickle')37 except Exception as e:38 print(f"Warning: Failed to load punkt tokenizer: {str(e)}")39 # Fallback to default sent_tokenize40 self.tokenizer = nltk.tokenize.sent_tokenize41 42 def summarize_text(self, text: str) -> Dict:43 """텍스트를 요약"""44 try:45 # 텍스트 분할46 chunks = self._split_text(text)47 48 # 각 분할에 대해 요약 생성49 summaries = []50 for chunk in chunks:51 summary = self._summarize_chunk(chunk)52 if summary:53 summaries.append(summary)54 55 return {56 "timestamp": datetime.now().isoformat(),57 "full_summary": " ".join(summaries),58 "chunk_summaries": summaries59 }60 61 except Exception as e:62 raise Exception(f"요약 생성 중 오류 발생: {str(e)}")63 64 def _summarize_chunk(self, text: str) -> str:65 """개별 텍스트 분할을 요약"""66 try:67 # 텍스트 전처리68 words = nltk.word_tokenize(text.lower())69 sentences = nltk.sent_tokenize(text)70 71 # 불용어 제거72 stop_words = set(nltk.corpus.stopwords.words('english'))73 words = [word for word in words if word.isalnum() and word not in stop_words]74 75 # 단어 빈도수 계산76 word_frequencies = {}77 for word in words:78 if word not in word_frequencies:79 word_frequencies[word] = 180 else:81 word_frequencies[word] += 182 83 # 최대 빈도수 계산84 max_frequency = max(word_frequencies.values())85 86 # 정규화된 빈도수 계산87 for word in word_frequencies:88 word_frequencies[word] = word_frequencies[word] / max_frequency89 90 # 문장 점수 계산91 sentence_scores = {}92 for sentence in sentences:93 for word, freq in word_frequencies.items():94 if word in sentence.lower():95 if sentence not in sentence_scores:96 sentence_scores[sentence] = freq97 else:98 sentence_scores[sentence] += freq99 100 # 상위 30%의 문장 선택101 summary_sentences = heapq.nlargest(102 int(len(sentences) * 0.3),103 sentence_scores,104 key=sentence_scores.get105 )106 107 # 요약 생성108 return " ".join(summary_sentences)109 110 except Exception as e:111 print(f"Chunk summarization error: {str(e)}")112 return ""113 114 def _split_text(self, text: str) -> List[str]:115 """텍스트를 적절한 크기로 분할"""116 try:117 # Use the configured tokenizer (either punkt or sent_tokenize)118 if hasattr(self, 'tokenizer') and callable(self.tokenizer):119 if self.tokenizer == nltk.tokenize.sent_tokenize:120 sentences = nltk.tokenize.sent_tokenize(text)121 else:122 # Handle the case where tokenizer is a PunktSentenceTokenizer instance123 sentences = self.tokenizer.tokenize(text)124 else:125 # Fallback to default sentence tokenizer126 nltk.download('punkt')127 sentences = nltk.tokenize.sent_tokenize(text)128 129 chunks = []130 current_chunk = ""131 132 for sentence in sentences:133 if len(current_chunk.split()) + len(sentence.split()) <= self.chunk_size:134 current_chunk = f"{current_chunk} {sentence}".strip()135 else:136 if current_chunk: # Only add non-empty chunks137 chunks.append(current_chunk)138 current_chunk = sentence139 140 # Add the last chunk if it's not empty141 if current_chunk:142 chunks.append(current_chunk.strip())143 144 return chunks if chunks else [text] # Return at least one chunk145 146 except LookupError as e:147 # If punkt data is missing, try to download it148 print(f"NLTK data missing, attempting to download: {e}")149 nltk.download('punkt')150 # Retry with the default tokenizer151 return self._split_text(text)152 except Exception as e:153 print(f"Error in _split_text: {str(e)}")154 # If all else fails, return the original text as a single chunk155 return [text]156 157# 싱글톤 인스턴스 생성158document_summarizer = DocumentSummarizer()159 