abprasadhuggingface/Hindi-BPE-Encoder-Decoder
0
1import streamlit as st2from bpe import HindiBPE3import json4import os5 6def load_bpe_model():7 """Load BPE model from saved file or train new one"""8 model_path = "data/bpe_model.json"9 if os.path.exists(model_path):10 bpe = HindiBPE()11 bpe.load_model(model_path)12 return bpe13 else:14 from download_data import download_hindi_corpus15 from preprocessor import load_and_preprocess_data16 17 corpus_path = download_hindi_corpus()18 processed_texts = load_and_preprocess_data(corpus_path)19 20 bpe = HindiBPE(vocab_size=50000)21 bpe.fit(processed_texts[:1000])22 bpe.save_model()23 return bpe24 25def format_tokens(indices: list, token_mapping: dict) -> str:26 """Format indices with their corresponding tokens"""27 return ' '.join([f"{idx}({token_mapping[idx]})" for idx in indices])28 29def save_encoded_result(bpe_model, text: str, indices: list):30 """Save encoded text and its tokens"""31 bpe_model.save_encoded_text(text, indices)32 33def load_encoded_texts():34 """Load previously encoded texts"""35 path = "data/encoded_texts.json"36 if os.path.exists(path):37 with open(path, 'r', encoding='utf-8') as f:38 return json.load(f)39 return {'texts': []}40 41def main():42 st.title("हिंदी BPE एनकोडर/डिकोडर (Hindi BPE Encoder/Decoder)")43 44 # Initialize BPE model45 if 'bpe_model' not in st.session_state:46 with st.spinner('BPE मॉडल लोड हो रहा है... (Loading BPE model...)'):47 st.session_state.bpe_model = load_bpe_model()48 49 # Create tabs for encode, decode, and history50 tab1, tab2, tab3 = st.tabs(["एनकोड (Encode)", "डिकोड (Decode)", "इतिहास (History)"])51 52 with tab1:53 st.subheader("टेक्स्ट एनकोड करें (Encode Text)")54 55 # Input text area56 input_text = st.text_area(57 "हिंदी टेक्स्ट दर्ज करें (Enter Hindi text):",58 height=100,59 placeholder="यहाँ टेक्स्ट टाइप करें..."60 )61 62 if st.button("एनकोड करें (Encode)"):63 if input_text:64 # Encode the text65 indices = st.session_state.bpe_model.encode(input_text)66 token_mapping = st.session_state.bpe_model.get_token_mapping()67 68 # Save the encoded result69 save_encoded_result(st.session_state.bpe_model, input_text, indices)70 71 # Display results72 st.subheader("एनकोडेड टोकन (Encoded Tokens):")73 74 # Display indices only75 st.text_area(76 "इंडेक्स (Indices):", 77 ' '.join(map(str, indices)), 78 height=7079 )80 81 # Display indices with corresponding tokens82 formatted_tokens = format_tokens(indices, token_mapping)83 st.text_area(84 "इंडेक्स और टोकन (Indices with Tokens):", 85 formatted_tokens, 86 height=10087 )88 89 # Add copy buttons90 st.markdown("### कॉपी करें (Copy):")91 col1, col2 = st.columns(2)92 93 # JSON format for copying94 indices_json = json.dumps(indices)95 with col1:96 st.code(indices_json, language='json')97 if st.button("JSON कॉपी करें", key="copy_json"):98 st.write("JSON copied!")99 100 with col2:101 st.code(' '.join(map(str, indices)), language='text')102 if st.button("टेक्स्ट कॉपी करें", key="copy_text"):103 st.write("Text copied!")104 105 # Display statistics106 st.subheader("सांख्यिकी (Statistics):")107 st.write(f"मूल लंबाई (Original length): {len(input_text)} characters")108 st.write(f"टोकन की संख्या (Number of tokens): {len(indices)}")109 st.write(f"कम्प्रेशन अनुपात (Compression ratio): {len(input_text)/len(indices):.2f}x")110 111 with tab2:112 st.subheader("टोकन डिकोड करें (Decode Tokens)")113 114 # Input format selection115 input_format = st.radio(116 "इनपुट फॉर्मैट चुनें (Select input format):",117 ["JSON", "Space-separated numbers"]118 )119 120 # Input area for indices121 indices_input = st.text_area(122 "इंडेक्स दर्ज करें (Enter indices):",123 height=100,124 placeholder="JSON या स्पेस-सेपरेटेड नंबर्स दर्ज करें..."125 )126 127 if st.button("डिकोड करें (Decode)"):128 if indices_input:129 try:130 # Parse input based on format131 if input_format == "JSON":132 indices = json.loads(indices_input)133 else:134 indices = [int(idx) for idx in indices_input.strip().split()]135 136 # Decode indices137 decoded_text = st.session_state.bpe_model.decode(indices)138 139 # Display result140 st.subheader("डिकोडेड टेक्स्ट (Decoded Text):")141 st.text_area(142 "परिणाम (Result):", 143 decoded_text, 144 height=100145 )146 147 except Exception as e:148 st.error(f"Error: {str(e)}")149 150 with tab3:151 st.subheader("एनकोडेड टेक्स्ट इतिहास (Encoded Text History)")152 153 # Load and display encoded history154 encoded_history = load_encoded_texts()155 156 if encoded_history['texts']:157 for idx, entry in enumerate(reversed(encoded_history['texts'])):158 with st.expander(f"Text {idx + 1}: {entry['original_text'][:50]}..."):159 st.write("Original Text:")160 st.write(entry['original_text'])161 st.write("Encoded Indices:")162 st.code(json.dumps(entry['indices']))163 st.write("Tokens:")164 st.write(entry['token_mappings'])165 else:166 st.write("No encoded texts in history.")167 168if __name__ == "__main__":169 main() 