Manmeet2002/Accessibility_Project
0
1import os2import base643import json4import ast # For safely evaluating agent string outputs5from openai import OpenAI6from crewai import Agent, Task, Crew, Process7from crewai.tools import BaseTool8from PIL import Image9import streamlit as st10 11# --- 1. API KEY SETUP ---12# Streamlit will get the key from .streamlit/secrets.toml13try:14 OPENAI_API_KEY = st.secrets["OPENAI_API_KEY"]15 os.environ['OPENAI_API_KEY'] = OPENAI_API_KEY16 client = OpenAI(api_key=OPENAI_API_KEY)17except KeyError:18 st.error("OPENAI_API_KEY not found! Please add it to your .streamlit/secrets.toml file.")19 st.stop()20 21 22# --- 2. HELPER FUNCTION & ALL TOOL DEFINITIONS ---23 24# Helper for encoding images25def encode_image(image_path):26 with open(image_path, "rb") as image_file:27 return base64.b64encode(image_file.read()).decode('utf-8')28 29# --- Module 1 Tools ---30class VisualTools(BaseTool):31 name: str = "Image Description Tool"32 description: str = "Takes the file path of an image and generates brief, standard, and detailed descriptions using GPT-4o vision."33 34 def _run(self, image_path: str) -> str:35 base64_image = encode_image(image_path)36 try:37 response = client.chat.completions.create(38 model="gpt-4o",39 messages=[40 {"role": "user", "content": [41 {"type": "text", "text": "Describe this image in detail. Follow W3C WAI guidelines. Provide three versions: a 'Brief' description, a 'Standard' description, and a 'Detailed' description."},42 {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{base64_image}"}}43 ]}44 ], max_tokens=50045 )46 return response.choices[0].message.content47 except Exception as e:48 return f"Error analyzing image: {e}"49 50class AudioTools(BaseTool):51 name: str = "Text-to-Speech Tool"52 description: str = "Takes text and a save path, converts it to an MP3, and saves it."53 54 def _run(self, text_to_speak: str, save_path: str = "module_1_output.mp3") -> str:55 try:56 audio_response = client.audio.speech.create(57 model="tts-1", voice="nova", input=text_to_speak58 )59 audio_response.stream_to_file(save_path)60 return f"Audio file saved to {save_path}"61 except Exception as e:62 return f"Error generating audio: {e}"63 64# --- Module 2 Tool ---65class VisualSimplifierTool(BaseTool):66 name: str = "Complex Text to Visual Tool"67 description: str = "Takes complex text and generates a text-free visual explanation. Returns the URL."68 69 def _run(self, complex_text: str) -> str:70 try:71 simplification_prompt = f"""72 You are a visual learning designer. Read the following complex text.73 Your goal is to generate a DALL-E 3 prompt to create a simple, clear74 visual diagram or infographic that explains this concept.75 ***IMPORTANT RULE: The visual must contain NO TEXT, NO LABELS, and NO LETTERS.***76 The prompt should describe a visual metaphor or a diagram using only icons and arrows.77 78 Complex Text: "{complex_text}"79 80 Respond with *only* the DALL-E 3 prompt and nothing else.81 """82 83 response = client.chat.completions.create(84 model="gpt-4o", messages=[{"role": "user", "content": simplification_prompt}], max_tokens=30085 )86 dalle_prompt = response.choices[0].message.content87 88 image_response = client.images.generate(89 model="dall-e-3", prompt=dalle_prompt, size="1024x1024", quality="standard", n=190 )91 return image_response.data[0].url92 except Exception as e:93 return f"Error in visual generation process: {e}"94 95# --- Module 4 Tools ---96class ASLGrammarTool(BaseTool):97 name: str = "ASL Grammar Conversion Tool"98 description: str = "Takes an English sentence and converts it into the correct ASL grammatical structure (e.g., Topic-Comment)."99 100 def _run(self, english_text: str) -> str:101 system_prompt = """102 You are an expert ASL linguist. Your job is to translate English sentences103 into the correct ASL grammatical structure.104 Key rules: Topic-Comment, No Articles, No 'be' verbs, Use Infinitives, Time-first.105 Example:106 - English: "I am going to the store tomorrow."107 - ASL Grammar: "TOMORROW, STORE, I GO-TO."108 Respond with *only* the ASL grammar string and nothing else.109 """110 try:111 response = client.chat.completions.create(112 model="gpt-4o",113 messages=[114 {"role": "system", "content": system_prompt},115 {"role": "user", "content": english_text}116 ], max_tokens=100117 )118 asl_grammar_string = response.choices[0].message.content119 # Standardize output for the next agent120 signs = [sign.strip().upper() for sign in asl_grammar_string.split(',')]121 return signs122 except Exception as e:123 return f"Error in grammar conversion: {e}"124 125# Load our new local video map ONCE when the app starts126try:127 with open('sign_video_map.json', 'r') as f:128 SIGN_VIDEO_MAP = json.load(f)129 print("Local sign video map loaded successfully.")130except FileNotFoundError:131 print("ERROR: sign_video_map.json not found. Did you run parse_data.py first?")132 SIGN_VIDEO_MAP = {}133 134class SignDetailTool(BaseTool):135 name: str = "Sign Detail Lookup Tool"136 description: str = "Takes a Python LIST of ASL sign words and returns a list of dictionaries. Each dictionary contains the sign, its video path, and its text description."137 138 def _run(self, sign_list: list) -> list:139 print(f"--- Tool: Looking up details (video + text) for {sign_list}... ---")140 141 details_list = []142 for sign_word in sign_list:143 clean_word = sign_word.lower().strip().replace("-", "")144 145 # 1. Get Video Path (from our JSON map)146 video_id = SIGN_VIDEO_MAP.get(clean_word, None)147 video_path = f"videos/{video_id}.mp4" if video_id else f"No video found for sign: {sign_word}"148 149 # 2. Get Text Description (from AI)150 description = "No description found." # Default151 try:152 system_prompt = f"""153 You are a sign language dictionary. Given a single ASL sign word, provide a brief,154 one-sentence description of how to perform the sign.155 Example: - Input: "STORE" - Output: "With both hands in 'S' shape, tap fingertips together twice."156 157 Input: "{sign_word}"158 Output:159 """160 response = client.chat.completions.create(161 model="gpt-4o", messages=[{"role": "system", "content": system_prompt}], max_tokens=100162 )163 description = response.choices[0].message.content.strip()164 except Exception as e:165 description = f"Error getting description: {e}"166 167 # 3. Append both to our list168 details_list.append({169 "sign": sign_word,170 "path": video_path,171 "desc": description172 })173 174 return details_list175 176# --- 3. INSTANTIATE ALL TOOLS ---177visual_tool = VisualTools()178audio_tool = AudioTools()179visual_simplifier_tool = VisualSimplifierTool()180grammar_tool = ASLGrammarTool()181detail_tool = SignDetailTool() # <-- Note: new tool instantiated182 183# --- 4. DEFINE ALL AGENTS ---184 185# Module 1 Agents186visual_describer = Agent(187 role='Visual Describer',188 goal='Create brief, standard, and detailed accessibility descriptions for an image.',189 backstory='You are an expert in web accessibility and image analysis.',190 tools=[visual_tool], verbose=False, allow_delegation=False191)192audio_producer = Agent(193 role='Audio Producer',194 goal='Convert text descriptions into natural, high-quality audio files.',195 backstory='You are a professional voice actor with a clear and engaging voice.',196 tools=[audio_tool], verbose=False, allow_delegation=False197)198 199# Module 2 Agent200visual_simplifier_agent = Agent(201 role='Visual Simplifier',202 goal='Create a clear and simple text-free visual explanation from complex text.',203 backstory='You are an expert in instructional design and cognitive science.',204 tools=[visual_simplifier_tool], verbose=False, allow_delegation=False205)206 207# Module 3 Agents208asl_grammar_agent = Agent(209 role='ASL Grammar Translator',210 goal='Convert spoken English sentences into the correct ASL grammatical sign order.',211 backstory='You are an expert linguist specializing in English to ASL translation.',212 tools=[grammar_tool], verbose=False, allow_delegation=False213)214sign_sequencer_agent = Agent(215 role='Sign Language Sequencer',216 goal='Take a list of ASL-grammar signs and find the local video file path AND text description for each one.',217 backstory='You are a sign language librarian. You take a list of signs and return a list of objects containing all details for each sign.',218 tools=[detail_tool], # <-- THIS IS THE CHANGE219 verbose=False,220 allow_delegation=False221)222 223# --- 5. STREAMLIT WEB INTERFACE ---224st.title("Multimodal Accessibility Translator ๐ค")225st.markdown("This app uses a CrewAI multi-agent system to make content accessible.")226 227tab1, tab2, tab3 = st.tabs([228 "๐๏ธ Module 1: Image-to-Audio",229 "๐ง Module 2: Text-to-Visual",230 "โ Module 3: Text-to-Sign Language"231])232 233# --- TAB 1: Image-to-Audio ---234with tab1:235 st.header("Generate Audio Descriptions from an Image")236 uploaded_file = st.file_uploader("Choose an image...", type=["jpg", "jpeg", "png"])237 238 if uploaded_file is not None:239 # Save the uploaded file to a temporary path240 img = Image.open(uploaded_file)241 temp_image_path = "temp_uploaded_image.jpg"242 img.save(temp_image_path)243 244 # --- FIX: Changed 'use_column_width' to 'use_container_width' ---245 st.image(img, caption='Uploaded Image.', use_container_width=True)246 247 if st.button("Generate Accessibility Audio", key="mod1_go"):248 with st.spinner("Agents are working... This may take a moment."):249 250 # Define tasks251 description_task = Task(252 description=f'Analyze the image at {temp_image_path} and generate descriptions.',253 expected_output='A single string with brief, standard, and detailed descriptions.',254 agent=visual_describer255 )256 audio_task = Task(257 description='Convert the text descriptions into a single audio file named "module_1_output.mp3".',258 expected_output='A confirmation string that the audio file was saved.',259 agent=audio_producer,260 context=[description_task]261 )262 263 # Define crew264 accessibility_crew = Crew(265 agents=[visual_describer, audio_producer],266 tasks=[description_task, audio_task],267 process=Process.sequential,268 verbose=False269 )270 271 # Run crew272 crew_result = accessibility_crew.kickoff()273 274 # --- FIX: Get the text description from crew_result ---275 text_description = crew_result.tasks_output[0].raw276 277 st.subheader("Generated Descriptions:")278 st.markdown(text_description)279 280 st.subheader("Generated Audio:")281 st.audio("module_1_output.mp3")282 283 # Clean up temp file284 os.remove(temp_image_path)285 286# --- TAB 2: Text-to-Visual ---287with tab2:288 st.header("Generate a Text-Free Visual from Complex Text")289 complex_text = st.text_area("Paste your complex text here:", height=150, 290 placeholder="e.g., Photosynthesis is a process used by plants...")291 292 if st.button("Generate Visual Explanation", key="mod3_go"):293 if complex_text:294 with st.spinner("Visual Simplifier Agent is thinking..."):295 296 # Define task297 simplification_task = Task(298 description=f'Use your tool to convert the following complex text into a single, text-free visual explanation: "{complex_text}"',299 expected_output='The final URL of the generated visual diagram.',300 agent=visual_simplifier_agent301 )302 303 # Define crew304 visual_crew = Crew(305 agents=[visual_simplifier_agent],306 tasks=[simplification_task],307 process=Process.sequential,308 verbose=False309 )310 311 # Run crew312 crew_result = visual_crew.kickoff()313 314 st.subheader("Generated Visual (No Text):")315 if crew_result.raw and crew_result.raw.startswith('http'):316 st.image(crew_result.raw, caption="AI-generated visual explanation.")317 else:318 st.error(f"Agent failed to generate image. Log: {crew_result.raw}")319 else:320 st.warning("Please paste some text first.")321 322# --- TAB 3: Text-to-Sign Language ---323with tab3:324 st.header("Generate Sign Language Video from Text")325 english_text = st.text_input("Enter a simple English sentence:", 326 placeholder="e.g., book")327 328 if st.button("Generate Sign Sequence", key="mod2_go"):329 if english_text:330 with st.spinner("Sign Language 'Brain' Crew is working... (This may take a moment)"):331 332 # Define tasks333 grammar_task = Task(334 description=f"Translate the following English sentence into ASL grammar: '{english_text}'",335 expected_output="A Python list of ASL sign words in the correct grammatical order.",336 agent=asl_grammar_agent337 )338 339 sequence_task = Task(340 description="You have a Python list of ASL signs. Pass this *entire list* to your 'Sign Detail Lookup Tool' to get the final list of details.",341 expected_output="A final Python list of dictionaries, where each dictionary contains 'sign', 'path', and 'desc'.",342 agent=sign_sequencer_agent,343 context=[grammar_task] 344 )345 346 # Define crew347 sign_language_crew = Crew(348 agents=[asl_grammar_agent, sign_sequencer_agent],349 tasks=[grammar_task, sequence_task],350 process=Process.sequential,351 verbose=False352 )353 354 # Run crew355 crew_result = sign_language_crew.kickoff()356 357 st.subheader("Generated Sign Language Instructions & Videos:")358 359 if crew_result.raw:360 try:361 # --- FIX: Convert string-list to real list ---362 details_list = ast.literal_eval(crew_result.raw)363 364 if not isinstance(details_list, list):365 st.error(f"Agent returned unexpected data: {details_list}")366 st.stop()367 368 # Create columns for a cleaner layout369 if not details_list:370 st.warning("No signs were generated. Try a different sentence.")371 st.stop()372 373 cols = st.columns(len(details_list))374 375 for i, item in enumerate(details_list):376 with cols[i]:377 # 1. Show the Sign378 st.markdown(f"**{item['sign']}**")379 # 2. Show the Text Description380 st.markdown(f"*{item['desc']}*") 381 382 video_path = item['path']383 # 3. Show the Video384 if video_path.startswith('videos/') and os.path.exists(video_path):385 try:386 st.video(video_path)387 except Exception as e:388 st.error(f"Error loading video: {e}")389 elif video_path.startswith('videos/'):390 st.error(f"File not found: {video_path}")391 else:392 st.error(video_path) # Show "No video found..."393 394 except Exception as e:395 st.error(f"Failed to process agent output. Error: {e}")396 st.text(f"Raw output: {crew_config.raw}")397 398 else:399 st.error(f"Agent failed to generate sequence. Log: {crew_result}")400 else:401 st.warning("Please enter a sentence first.") y=alt.Y("y", axis=None),402 color=alt.Color("idx", legend=None, scale=alt.Scale()),403 size=alt.Size("rand", legend=None, scale=alt.Scale(range=[1, 150])),404 ))