abprasadhuggingface/Hindi-BPE-Encoder-Decoder
0
1import re2from typing import List3import xml.etree.ElementTree as ET4 5def clean_wiki_text(text: str) -> str:6 """Remove Wikipedia markup and extract plain text"""7 # Remove XML tags8 text = re.sub(r'<[^>]+>', ' ', text)9 10 # Remove Wikipedia markup11 text = re.sub(r'\{\{[^\}]+\}\}', ' ', text)12 text = re.sub(r'\[\[[^\]]+\]\]', ' ', text)13 text = re.sub(r'\{\|[^\}]+\|\}', ' ', text)14 15 return text16 17def preprocess_hindi_text(text: str) -> str:18 """Preprocess Hindi text by removing unnecessary characters and normalizing"""19 20 # Clean Wikipedia markup21 text = clean_wiki_text(text)22 23 # Remove URLs24 text = re.sub(r'http\S+|www.\S+', '', text)25 26 # Remove English characters and numbers27 text = re.sub(r'[A-Za-z0-9]', '', text)28 29 # Remove extra whitespace30 text = re.sub(r'\s+', ' ', text)31 32 # Remove special characters except Hindi-specific ones33 text = re.sub(r'[^\u0900-\u097F\s]', '', text)34 35 return text.strip()36 37def load_and_preprocess_data(file_path: str) -> List[str]:38 """Load and preprocess the Hindi corpus"""39 40 with open(file_path, 'r', encoding='utf-8') as f:41 content = f.read()42 43 # Split into sentences (roughly)44 texts = re.split(r'[।\n]', content)45 46 # Preprocess each line47 processed_texts = [preprocess_hindi_text(text) for text in texts]48 49 # Remove empty lines50 processed_texts = [text for text in processed_texts if text.strip()]51 52 return processed_texts 