Team Ai
Apppublic

abprasadhuggingface/Hindi-BPE-Encoder-Decoder

sourceHugging Faceupdated 2y agoView on Hugging Face
0likes
app.py169 linesDownload Raw Back to root
1import streamlit as st2from bpe import HindiBPE3import json4import os5 6def load_bpe_model():7    """Load BPE model from saved file or train new one"""8    model_path = "data/bpe_model.json"9    if os.path.exists(model_path):10        bpe = HindiBPE()11        bpe.load_model(model_path)12        return bpe13    else:14        from download_data import download_hindi_corpus15        from preprocessor import load_and_preprocess_data16        17        corpus_path = download_hindi_corpus()18        processed_texts = load_and_preprocess_data(corpus_path)19        20        bpe = HindiBPE(vocab_size=50000)21        bpe.fit(processed_texts[:1000])22        bpe.save_model()23        return bpe24 25def format_tokens(indices: list, token_mapping: dict) -> str:26    """Format indices with their corresponding tokens"""27    return ' '.join([f"{idx}({token_mapping[idx]})" for idx in indices])28 29def save_encoded_result(bpe_model, text: str, indices: list):30    """Save encoded text and its tokens"""31    bpe_model.save_encoded_text(text, indices)32 33def load_encoded_texts():34    """Load previously encoded texts"""35    path = "data/encoded_texts.json"36    if os.path.exists(path):37        with open(path, 'r', encoding='utf-8') as f:38            return json.load(f)39    return {'texts': []}40 41def main():42    st.title("हिंदी BPE एनकोडर/डिकोडर (Hindi BPE Encoder/Decoder)")43    44    # Initialize BPE model45    if 'bpe_model' not in st.session_state:46        with st.spinner('BPE मॉडल लोड हो रहा है... (Loading BPE model...)'):47            st.session_state.bpe_model = load_bpe_model()48    49    # Create tabs for encode, decode, and history50    tab1, tab2, tab3 = st.tabs(["एनकोड (Encode)", "डिकोड (Decode)", "इतिहास (History)"])51    52    with tab1:53        st.subheader("टेक्स्ट एनकोड करें (Encode Text)")54        55        # Input text area56        input_text = st.text_area(57            "हिंदी टेक्स्ट दर्ज करें (Enter Hindi text):",58            height=100,59            placeholder="यहाँ टेक्स्ट टाइप करें..."60        )61        62        if st.button("एनकोड करें (Encode)"):63            if input_text:64                # Encode the text65                indices = st.session_state.bpe_model.encode(input_text)66                token_mapping = st.session_state.bpe_model.get_token_mapping()67                68                # Save the encoded result69                save_encoded_result(st.session_state.bpe_model, input_text, indices)70                71                # Display results72                st.subheader("एनकोडेड टोकन (Encoded Tokens):")73                74                # Display indices only75                st.text_area(76                    "इंडेक्स (Indices):", 77                    ' '.join(map(str, indices)), 78                    height=7079                )80                81                # Display indices with corresponding tokens82                formatted_tokens = format_tokens(indices, token_mapping)83                st.text_area(84                    "इंडेक्स और टोकन (Indices with Tokens):", 85                    formatted_tokens, 86                    height=10087                )88                89                # Add copy buttons90                st.markdown("### कॉपी करें (Copy):")91                col1, col2 = st.columns(2)92                93                # JSON format for copying94                indices_json = json.dumps(indices)95                with col1:96                    st.code(indices_json, language='json')97                    if st.button("JSON कॉपी करें", key="copy_json"):98                        st.write("JSON copied!")99                100                with col2:101                    st.code(' '.join(map(str, indices)), language='text')102                    if st.button("टेक्स्ट कॉपी करें", key="copy_text"):103                        st.write("Text copied!")104                105                # Display statistics106                st.subheader("सांख्यिकी (Statistics):")107                st.write(f"मूल लंबाई (Original length): {len(input_text)} characters")108                st.write(f"टोकन की संख्या (Number of tokens): {len(indices)}")109                st.write(f"कम्प्रेशन अनुपात (Compression ratio): {len(input_text)/len(indices):.2f}x")110    111    with tab2:112        st.subheader("टोकन डिकोड करें (Decode Tokens)")113        114        # Input format selection115        input_format = st.radio(116            "इनपुट फॉर्मैट चुनें (Select input format):",117            ["JSON", "Space-separated numbers"]118        )119        120        # Input area for indices121        indices_input = st.text_area(122            "इंडेक्स दर्ज करें (Enter indices):",123            height=100,124            placeholder="JSON या स्पेस-सेपरेटेड नंबर्स दर्ज करें..."125        )126        127        if st.button("डिकोड करें (Decode)"):128            if indices_input:129                try:130                    # Parse input based on format131                    if input_format == "JSON":132                        indices = json.loads(indices_input)133                    else:134                        indices = [int(idx) for idx in indices_input.strip().split()]135                    136                    # Decode indices137                    decoded_text = st.session_state.bpe_model.decode(indices)138                    139                    # Display result140                    st.subheader("डिकोडेड टेक्स्ट (Decoded Text):")141                    st.text_area(142                        "परिणाम (Result):", 143                        decoded_text, 144                        height=100145                    )146                    147                except Exception as e:148                    st.error(f"Error: {str(e)}")149    150    with tab3:151        st.subheader("एनकोडेड टेक्स्ट इतिहास (Encoded Text History)")152        153        # Load and display encoded history154        encoded_history = load_encoded_texts()155        156        if encoded_history['texts']:157            for idx, entry in enumerate(reversed(encoded_history['texts'])):158                with st.expander(f"Text {idx + 1}: {entry['original_text'][:50]}..."):159                    st.write("Original Text:")160                    st.write(entry['original_text'])161                    st.write("Encoded Indices:")162                    st.code(json.dumps(entry['indices']))163                    st.write("Tokens:")164                    st.write(entry['token_mappings'])165        else:166            st.write("No encoded texts in history.")167 168if __name__ == "__main__":169    main()