Team Ai
Apppublic

bluewhale2025/parseai-document-processor

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
summarizer.py159 linesDownload Raw Back to root
1import nltk2from typing import Dict, List3import json4from datetime import datetime5import heapq6 7class DocumentSummarizer:8    def __init__(self):9        # Set NLTK data path10        nltk_data_paths = [11            '/usr/local/share/nltk_data',12            '/usr/share/nltk_data',13            '/usr/local/nltk_data',14            '/usr/local/lib/nltk_data',15            '/usr/lib/nltk_data',16            '/root/nltk_data',17            '/home/user/nltk_data',18            '/app/nltk_data'19        ]20        21        # Add all possible NLTK data paths22        nltk.data.path = list(dict.fromkeys(nltk_data_paths + nltk.data.path))23        24        # Download NLTK data if not found25        try:26            nltk.download('punkt')27            nltk.download('stopwords')28            nltk.download('wordnet')29            nltk.download('averaged_perceptron_tagger')30        except Exception as e:31            print(f"Warning: NLTK data download failed: {str(e)}")32            33        # 텍스트 분할 크기 설정34        self.chunk_size = 1000  # 토큰 기준35        try:36            self.tokenizer = nltk.data.load('tokenizers/punkt/english.pickle')37        except Exception as e:38            print(f"Warning: Failed to load punkt tokenizer: {str(e)}")39            # Fallback to default sent_tokenize40            self.tokenizer = nltk.tokenize.sent_tokenize41        42    def summarize_text(self, text: str) -> Dict:43        """텍스트를 요약"""44        try:45            # 텍스트 분할46            chunks = self._split_text(text)47            48            # 각 분할에 대해 요약 생성49            summaries = []50            for chunk in chunks:51                summary = self._summarize_chunk(chunk)52                if summary:53                    summaries.append(summary)54            55            return {56                "timestamp": datetime.now().isoformat(),57                "full_summary": " ".join(summaries),58                "chunk_summaries": summaries59            }60            61        except Exception as e:62            raise Exception(f"요약 생성 중 오류 발생: {str(e)}")63 64    def _summarize_chunk(self, text: str) -> str:65        """개별 텍스트 분할을 요약"""66        try:67            # 텍스트 전처리68            words = nltk.word_tokenize(text.lower())69            sentences = nltk.sent_tokenize(text)70            71            # 불용어 제거72            stop_words = set(nltk.corpus.stopwords.words('english'))73            words = [word for word in words if word.isalnum() and word not in stop_words]74            75            # 단어 빈도수 계산76            word_frequencies = {}77            for word in words:78                if word not in word_frequencies:79                    word_frequencies[word] = 180                else:81                    word_frequencies[word] += 182            83            # 최대 빈도수 계산84            max_frequency = max(word_frequencies.values())85            86            # 정규화된 빈도수 계산87            for word in word_frequencies:88                word_frequencies[word] = word_frequencies[word] / max_frequency89            90            # 문장 점수 계산91            sentence_scores = {}92            for sentence in sentences:93                for word, freq in word_frequencies.items():94                    if word in sentence.lower():95                        if sentence not in sentence_scores:96                            sentence_scores[sentence] = freq97                        else:98                            sentence_scores[sentence] += freq99            100            # 상위 30%의 문장 선택101            summary_sentences = heapq.nlargest(102                int(len(sentences) * 0.3),103                sentence_scores,104                key=sentence_scores.get105            )106            107            # 요약 생성108            return " ".join(summary_sentences)109            110        except Exception as e:111            print(f"Chunk summarization error: {str(e)}")112            return ""113    114    def _split_text(self, text: str) -> List[str]:115        """텍스트를 적절한 크기로 분할"""116        try:117            # Use the configured tokenizer (either punkt or sent_tokenize)118            if hasattr(self, 'tokenizer') and callable(self.tokenizer):119                if self.tokenizer == nltk.tokenize.sent_tokenize:120                    sentences = nltk.tokenize.sent_tokenize(text)121                else:122                    # Handle the case where tokenizer is a PunktSentenceTokenizer instance123                    sentences = self.tokenizer.tokenize(text)124            else:125                # Fallback to default sentence tokenizer126                nltk.download('punkt')127                sentences = nltk.tokenize.sent_tokenize(text)128            129            chunks = []130            current_chunk = ""131            132            for sentence in sentences:133                if len(current_chunk.split()) + len(sentence.split()) <= self.chunk_size:134                    current_chunk = f"{current_chunk} {sentence}".strip()135                else:136                    if current_chunk:  # Only add non-empty chunks137                        chunks.append(current_chunk)138                    current_chunk = sentence139            140            # Add the last chunk if it's not empty141            if current_chunk:142                chunks.append(current_chunk.strip())143            144            return chunks if chunks else [text]  # Return at least one chunk145            146        except LookupError as e:147            # If punkt data is missing, try to download it148            print(f"NLTK data missing, attempting to download: {e}")149            nltk.download('punkt')150            # Retry with the default tokenizer151            return self._split_text(text)152        except Exception as e:153            print(f"Error in _split_text: {str(e)}")154            # If all else fails, return the original text as a single chunk155            return [text]156 157# 싱글톤 인스턴스 생성158document_summarizer = DocumentSummarizer()159