Team Ai
Apppublic

abprasadhuggingface/Hindi-BPE-Encoder-Decoder

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
preprocessor.py52 linesDownload Raw Back to root
1import re2from typing import List3import xml.etree.ElementTree as ET4 5def clean_wiki_text(text: str) -> str:6    """Remove Wikipedia markup and extract plain text"""7    # Remove XML tags8    text = re.sub(r'<[^>]+>', ' ', text)9    10    # Remove Wikipedia markup11    text = re.sub(r'\{\{[^\}]+\}\}', ' ', text)12    text = re.sub(r'\[\[[^\]]+\]\]', ' ', text)13    text = re.sub(r'\{\|[^\}]+\|\}', ' ', text)14    15    return text16 17def preprocess_hindi_text(text: str) -> str:18    """Preprocess Hindi text by removing unnecessary characters and normalizing"""19    20    # Clean Wikipedia markup21    text = clean_wiki_text(text)22    23    # Remove URLs24    text = re.sub(r'http\S+|www.\S+', '', text)25    26    # Remove English characters and numbers27    text = re.sub(r'[A-Za-z0-9]', '', text)28    29    # Remove extra whitespace30    text = re.sub(r'\s+', ' ', text)31    32    # Remove special characters except Hindi-specific ones33    text = re.sub(r'[^\u0900-\u097F\s]', '', text)34    35    return text.strip()36 37def load_and_preprocess_data(file_path: str) -> List[str]:38    """Load and preprocess the Hindi corpus"""39    40    with open(file_path, 'r', encoding='utf-8') as f:41        content = f.read()42    43    # Split into sentences (roughly)44    texts = re.split(r'[।\n]', content)45    46    # Preprocess each line47    processed_texts = [preprocess_hindi_text(text) for text in texts]48    49    # Remove empty lines50    processed_texts = [text for text in processed_texts if text.strip()]51    52    return processed_texts