Team Ai
Apppublic

Manmeet2002/Accessibility_Project

sourceHugging Faceupdated 11mo agoView on Hugging Face
0likes
streamlit_app.py404 linesDownload Raw Back to src
1import os2import base643import json4import ast  # For safely evaluating agent string outputs5from openai import OpenAI6from crewai import Agent, Task, Crew, Process7from crewai.tools import BaseTool8from PIL import Image9import streamlit as st10 11# --- 1. API KEY SETUP ---12# Streamlit will get the key from .streamlit/secrets.toml13try:14    OPENAI_API_KEY = st.secrets["OPENAI_API_KEY"]15    os.environ['OPENAI_API_KEY'] = OPENAI_API_KEY16    client = OpenAI(api_key=OPENAI_API_KEY)17except KeyError:18    st.error("OPENAI_API_KEY not found! Please add it to your .streamlit/secrets.toml file.")19    st.stop()20 21 22# --- 2. HELPER FUNCTION & ALL TOOL DEFINITIONS ---23 24# Helper for encoding images25def encode_image(image_path):26    with open(image_path, "rb") as image_file:27        return base64.b64encode(image_file.read()).decode('utf-8')28 29# --- Module 1 Tools ---30class VisualTools(BaseTool):31    name: str = "Image Description Tool"32    description: str = "Takes the file path of an image and generates brief, standard, and detailed descriptions using GPT-4o vision."33 34    def _run(self, image_path: str) -> str:35        base64_image = encode_image(image_path)36        try:37            response = client.chat.completions.create(38                model="gpt-4o",39                messages=[40                    {"role": "user", "content": [41                        {"type": "text", "text": "Describe this image in detail. Follow W3C WAI guidelines. Provide three versions: a 'Brief' description, a 'Standard' description, and a 'Detailed' description."},42                        {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_image}"}}43                    ]}44                ], max_tokens=50045            )46            return response.choices[0].message.content47        except Exception as e:48            return f"Error analyzing image: {e}"49 50class AudioTools(BaseTool):51    name: str = "Text-to-Speech Tool"52    description: str = "Takes text and a save path, converts it to an MP3, and saves it."53 54    def _run(self, text_to_speak: str, save_path: str = "module_1_output.mp3") -> str:55        try:56            audio_response = client.audio.speech.create(57                model="tts-1", voice="nova", input=text_to_speak58            )59            audio_response.stream_to_file(save_path)60            return f"Audio file saved to {save_path}"61        except Exception as e:62            return f"Error generating audio: {e}"63 64# --- Module 2 Tool ---65class VisualSimplifierTool(BaseTool):66    name: str = "Complex Text to Visual Tool"67    description: str = "Takes complex text and generates a text-free visual explanation. Returns the URL."68 69    def _run(self, complex_text: str) -> str:70        try:71            simplification_prompt = f"""72            You are a visual learning designer. Read the following complex text.73            Your goal is to generate a DALL-E 3 prompt to create a simple, clear74            visual diagram or infographic that explains this concept.75            ***IMPORTANT RULE: The visual must contain NO TEXT, NO LABELS, and NO LETTERS.***76            The prompt should describe a visual metaphor or a diagram using only icons and arrows.77            78            Complex Text: "{complex_text}"79            80            Respond with *only* the DALL-E 3 prompt and nothing else.81            """82            83            response = client.chat.completions.create(84                model="gpt-4o", messages=[{"role": "user", "content": simplification_prompt}], max_tokens=30085            )86            dalle_prompt = response.choices[0].message.content87 88            image_response = client.images.generate(89                model="dall-e-3", prompt=dalle_prompt, size="1024x1024", quality="standard", n=190            )91            return image_response.data[0].url92        except Exception as e:93            return f"Error in visual generation process: {e}"94 95# --- Module 4 Tools ---96class ASLGrammarTool(BaseTool):97    name: str = "ASL Grammar Conversion Tool"98    description: str = "Takes an English sentence and converts it into the correct ASL grammatical structure (e.g., Topic-Comment)."99 100    def _run(self, english_text: str) -> str:101        system_prompt = """102        You are an expert ASL linguist. Your job is to translate English sentences103        into the correct ASL grammatical structure.104        Key rules: Topic-Comment, No Articles, No 'be' verbs, Use Infinitives, Time-first.105        Example:106        - English: "I am going to the store tomorrow."107        - ASL Grammar: "TOMORROW, STORE, I GO-TO."108        Respond with *only* the ASL grammar string and nothing else.109        """110        try:111            response = client.chat.completions.create(112                model="gpt-4o",113                messages=[114                    {"role": "system", "content": system_prompt},115                    {"role": "user", "content": english_text}116                ], max_tokens=100117            )118            asl_grammar_string = response.choices[0].message.content119            # Standardize output for the next agent120            signs = [sign.strip().upper() for sign in asl_grammar_string.split(',')]121            return signs122        except Exception as e:123            return f"Error in grammar conversion: {e}"124 125# Load our new local video map ONCE when the app starts126try:127    with open('sign_video_map.json', 'r') as f:128        SIGN_VIDEO_MAP = json.load(f)129    print("Local sign video map loaded successfully.")130except FileNotFoundError:131    print("ERROR: sign_video_map.json not found. Did you run parse_data.py first?")132    SIGN_VIDEO_MAP = {}133 134class SignDetailTool(BaseTool):135    name: str = "Sign Detail Lookup Tool"136    description: str = "Takes a Python LIST of ASL sign words and returns a list of dictionaries. Each dictionary contains the sign, its video path, and its text description."137 138    def _run(self, sign_list: list) -> list:139        print(f"--- Tool: Looking up details (video + text) for {sign_list}... ---")140        141        details_list = []142        for sign_word in sign_list:143            clean_word = sign_word.lower().strip().replace("-", "")144            145            # 1. Get Video Path (from our JSON map)146            video_id = SIGN_VIDEO_MAP.get(clean_word, None)147            video_path = f"videos/{video_id}.mp4" if video_id else f"No video found for sign: {sign_word}"148            149            # 2. Get Text Description (from AI)150            description = "No description found." # Default151            try:152                system_prompt = f"""153                You are a sign language dictionary. Given a single ASL sign word, provide a brief,154                one-sentence description of how to perform the sign.155                Example: - Input: "STORE" - Output: "With both hands in 'S' shape, tap fingertips together twice."156                157                Input: "{sign_word}"158                Output:159                """160                response = client.chat.completions.create(161                    model="gpt-4o", messages=[{"role": "system", "content": system_prompt}], max_tokens=100162                )163                description = response.choices[0].message.content.strip()164            except Exception as e:165                description = f"Error getting description: {e}"166 167            # 3. Append both to our list168            details_list.append({169                "sign": sign_word,170                "path": video_path,171                "desc": description172            })173        174        return details_list175 176# --- 3. INSTANTIATE ALL TOOLS ---177visual_tool = VisualTools()178audio_tool = AudioTools()179visual_simplifier_tool = VisualSimplifierTool()180grammar_tool = ASLGrammarTool()181detail_tool = SignDetailTool() # <-- Note: new tool instantiated182 183# --- 4. DEFINE ALL AGENTS ---184 185# Module 1 Agents186visual_describer = Agent(187    role='Visual Describer',188    goal='Create brief, standard, and detailed accessibility descriptions for an image.',189    backstory='You are an expert in web accessibility and image analysis.',190    tools=[visual_tool], verbose=False, allow_delegation=False191)192audio_producer = Agent(193    role='Audio Producer',194    goal='Convert text descriptions into natural, high-quality audio files.',195    backstory='You are a professional voice actor with a clear and engaging voice.',196    tools=[audio_tool], verbose=False, allow_delegation=False197)198 199# Module 2 Agent200visual_simplifier_agent = Agent(201    role='Visual Simplifier',202    goal='Create a clear and simple text-free visual explanation from complex text.',203    backstory='You are an expert in instructional design and cognitive science.',204    tools=[visual_simplifier_tool], verbose=False, allow_delegation=False205)206 207# Module 3 Agents208asl_grammar_agent = Agent(209    role='ASL Grammar Translator',210    goal='Convert spoken English sentences into the correct ASL grammatical sign order.',211    backstory='You are an expert linguist specializing in English to ASL translation.',212    tools=[grammar_tool], verbose=False, allow_delegation=False213)214sign_sequencer_agent = Agent(215    role='Sign Language Sequencer',216    goal='Take a list of ASL-grammar signs and find the local video file path AND text description for each one.',217    backstory='You are a sign language librarian. You take a list of signs and return a list of objects containing all details for each sign.',218    tools=[detail_tool], # <-- THIS IS THE CHANGE219    verbose=False,220    allow_delegation=False221)222 223# --- 5. STREAMLIT WEB INTERFACE ---224st.title("Multimodal Accessibility Translator ๐Ÿค–")225st.markdown("This app uses a CrewAI multi-agent system to make content accessible.")226 227tab1, tab2, tab3 = st.tabs([228    "๐Ÿ‘๏ธ Module 1: Image-to-Audio",229    "๐Ÿง  Module 2: Text-to-Visual",230    "โœ‹ Module 3: Text-to-Sign Language"231])232 233# --- TAB 1: Image-to-Audio ---234with tab1:235    st.header("Generate Audio Descriptions from an Image")236    uploaded_file = st.file_uploader("Choose an image...", type=["jpg", "jpeg", "png"])237    238    if uploaded_file is not None:239        # Save the uploaded file to a temporary path240        img = Image.open(uploaded_file)241        temp_image_path = "temp_uploaded_image.jpg"242        img.save(temp_image_path)243        244        # --- FIX: Changed 'use_column_width' to 'use_container_width' ---245        st.image(img, caption='Uploaded Image.', use_container_width=True)246        247        if st.button("Generate Accessibility Audio", key="mod1_go"):248            with st.spinner("Agents are working... This may take a moment."):249                250                # Define tasks251                description_task = Task(252                    description=f'Analyze the image at {temp_image_path} and generate descriptions.',253                    expected_output='A single string with brief, standard, and detailed descriptions.',254                    agent=visual_describer255                )256                audio_task = Task(257                    description='Convert the text descriptions into a single audio file named "module_1_output.mp3".',258                    expected_output='A confirmation string that the audio file was saved.',259                    agent=audio_producer,260                    context=[description_task]261                )262                263                # Define crew264                accessibility_crew = Crew(265                    agents=[visual_describer, audio_producer],266                    tasks=[description_task, audio_task],267                    process=Process.sequential,268                    verbose=False269                )270                271                # Run crew272                crew_result = accessibility_crew.kickoff()273                274                # --- FIX: Get the text description from crew_result ---275                text_description = crew_result.tasks_output[0].raw276                277                st.subheader("Generated Descriptions:")278                st.markdown(text_description)279                280                st.subheader("Generated Audio:")281                st.audio("module_1_output.mp3")282                283                # Clean up temp file284                os.remove(temp_image_path)285 286# --- TAB 2: Text-to-Visual ---287with tab2:288    st.header("Generate a Text-Free Visual from Complex Text")289    complex_text = st.text_area("Paste your complex text here:", height=150, 290                                placeholder="e.g., Photosynthesis is a process used by plants...")291    292    if st.button("Generate Visual Explanation", key="mod3_go"):293        if complex_text:294            with st.spinner("Visual Simplifier Agent is thinking..."):295                296                # Define task297                simplification_task = Task(298                    description=f'Use your tool to convert the following complex text into a single, text-free visual explanation: "{complex_text}"',299                    expected_output='The final URL of the generated visual diagram.',300                    agent=visual_simplifier_agent301                )302                303                # Define crew304                visual_crew = Crew(305                    agents=[visual_simplifier_agent],306                    tasks=[simplification_task],307                    process=Process.sequential,308                    verbose=False309                )310                311                # Run crew312                crew_result = visual_crew.kickoff()313                314                st.subheader("Generated Visual (No Text):")315                if crew_result.raw and crew_result.raw.startswith('http'):316                    st.image(crew_result.raw, caption="AI-generated visual explanation.")317                else:318                    st.error(f"Agent failed to generate image. Log: {crew_result.raw}")319        else:320            st.warning("Please paste some text first.")321 322# --- TAB 3: Text-to-Sign Language ---323with tab3:324    st.header("Generate Sign Language Video from Text")325    english_text = st.text_input("Enter a simple English sentence:", 326                                 placeholder="e.g., book")327    328    if st.button("Generate Sign Sequence", key="mod2_go"):329        if english_text:330            with st.spinner("Sign Language 'Brain' Crew is working... (This may take a moment)"):331                332                # Define tasks333                grammar_task = Task(334                    description=f"Translate the following English sentence into ASL grammar: '{english_text}'",335                    expected_output="A Python list of ASL sign words in the correct grammatical order.",336                    agent=asl_grammar_agent337                )338                339                sequence_task = Task(340                    description="You have a Python list of ASL signs. Pass this *entire list* to your 'Sign Detail Lookup Tool' to get the final list of details.",341                    expected_output="A final Python list of dictionaries, where each dictionary contains 'sign', 'path', and 'desc'.",342                    agent=sign_sequencer_agent,343                    context=[grammar_task] 344                )345                346                # Define crew347                sign_language_crew = Crew(348                    agents=[asl_grammar_agent, sign_sequencer_agent],349                    tasks=[grammar_task, sequence_task],350                    process=Process.sequential,351                    verbose=False352                )353                354                # Run crew355                crew_result = sign_language_crew.kickoff()356                357                st.subheader("Generated Sign Language Instructions & Videos:")358                359                if crew_result.raw:360                    try:361                        # --- FIX: Convert string-list to real list ---362                        details_list = ast.literal_eval(crew_result.raw)363                        364                        if not isinstance(details_list, list):365                             st.error(f"Agent returned unexpected data: {details_list}")366                             st.stop()367                        368                        # Create columns for a cleaner layout369                        if not details_list:370                             st.warning("No signs were generated. Try a different sentence.")371                             st.stop()372                             373                        cols = st.columns(len(details_list))374                        375                        for i, item in enumerate(details_list):376                            with cols[i]:377                                # 1. Show the Sign378                                st.markdown(f"**{item['sign']}**")379                                # 2. Show the Text Description380                                st.markdown(f"*{item['desc']}*") 381                                382                                video_path = item['path']383                                # 3. Show the Video384                                if video_path.startswith('videos/') and os.path.exists(video_path):385                                    try:386                                        st.video(video_path)387                                    except Exception as e:388                                        st.error(f"Error loading video: {e}")389                                elif video_path.startswith('videos/'):390                                     st.error(f"File not found: {video_path}")391                                else:392                                    st.error(video_path) # Show "No video found..."393 394                    except Exception as e:395                        st.error(f"Failed to process agent output. Error: {e}")396                        st.text(f"Raw output: {crew_config.raw}")397 398                else:399                    st.error(f"Agent failed to generate sequence. Log: {crew_result}")400        else:401            st.warning("Please enter a sentence first.")        y=alt.Y("y", axis=None),402        color=alt.Color("idx", legend=None, scale=alt.Scale()),403        size=alt.Size("rand", legend=None, scale=alt.Scale(range=[1, 150])),404    ))