SahilCodevally/codevally-vision-language-action
0
1"""2src/ui/layout.py — Gradio Blocks layout definition for the VLA Demo.3 4This module owns *only* the component hierarchy and wiring.5Styling → src/ui/styles.py6HTML bits → src/ui/components.py7Handlers → src/ui/handlers.py8Examples → src/ui/examples.py9"""10 11import gradio as gr12 13from src.ui.styles import CUSTOM_CSS14from src.ui.components import HEADER_HTML, FOOTER_HTML15from src.ui.handlers import process_pipeline16from src.ui.examples import get_demo_examples17 18 19def create_ui() -> gr.Blocks:20 """21 Build and return the Gradio Blocks interface.22 23 All sub-modules are imported locally so that this function can be24 called in isolation for testing without triggering heavy model loads.25 """26 demo_examples = get_demo_examples()27 28 with gr.Blocks(29 title="VLA — AI Vision Language Action Pipeline | Codevally",30 theme=gr.themes.Base(31 primary_hue="blue",32 secondary_hue="cyan",33 neutral_hue="slate",34 font=[gr.themes.GoogleFont("Inter"), "system-ui", "sans-serif"],35 ),36 css=CUSTOM_CSS,37 ) as demo:38 39 # ── Header ────────────────────────────────────────────────────────40 gr.HTML(HEADER_HTML)41 42 # ── Main layout ───────────────────────────────────────────────────43 with gr.Row(equal_height=False):44 45 # ── Left: Inputs ───────────────────────────────────────────46 with gr.Column(scale=1, min_width=300):47 gr.Markdown("<div class='cv-section-label'>📷 Input</div>")48 image_input = gr.Image(49 label="Upload Scene Image",50 type="numpy",51 height=280,52 elem_id="image_upload",53 )54 gr.Markdown(55 "<div class='cv-section-label' style='margin-top:12px;'>"56 "💬 Natural Language Command</div>"57 )58 command_input = gr.Textbox(59 label="Command",60 placeholder=(61 'e.g. "Pick up the defective board and place it in the red bin"'62 ),63 lines=3,64 show_label=False,65 )66 run_btn = gr.Button(67 "🚀 Run AI Pipeline",68 variant="primary",69 size="lg",70 )71 gr.Markdown(72 "<div class='cv-section-label' style='margin-top:14px;'>"73 "📊 Pipeline Status</div>"74 )75 status_output = gr.Markdown(76 value="*Ready — upload a scene image and enter an instruction.*",77 elem_classes=["status-box"],78 )79 80 # ── Right: Outputs ─────────────────────────────────────────81 with gr.Column(scale=2):82 with gr.Tabs(elem_id="result_tabs"):83 84 with gr.TabItem("🔍 Detection Results"):85 gr.Markdown(86 "_Grounding DINO detects scene objects based on your "87 "command text — no fixed class limitations._"88 )89 with gr.Row():90 detection_image = gr.Image(91 label="Detected Objects",92 type="numpy",93 height=320,94 )95 detections_json = gr.Code(96 label="Detection Data (JSON)",97 language="json",98 lines=14,99 )100 101 with gr.TabItem("🎯 Action Plan & Visualization"):102 gr.Markdown(103 "_The LLM interprets your command and maps it to detected "104 "objects, generating step-by-step robotic action instructions._"105 )106 with gr.Row():107 action_image = gr.Image(108 label="Action Visualization",109 type="numpy",110 height=320,111 )112 action_plan_md = gr.Markdown(113 value=(114 "*Run the pipeline to see the "115 "AI-generated action plan.*"116 ),117 )118 119 # ── Demo examples ──────────────────────────────────────────────────120 if demo_examples:121 gr.Markdown("---")122 gr.Markdown(123 "### 📂 Try a Demo Scene\n"124 "_Click any example below to load a scene image with a pre-filled "125 "command. These showcase Codevally's VLA pipeline across diverse "126 "industrial and everyday environments._"127 )128 gr.Examples(129 examples=demo_examples,130 inputs=[image_input, command_input],131 label="Demo Scenes",132 examples_per_page=5,133 )134 135 # ── Footer ────────────────────────────────────────────────────────136 gr.HTML(FOOTER_HTML)137 138 # ── Event wiring ──────────────────────────────────────────────────139 run_btn.click(140 fn=process_pipeline,141 inputs=[image_input, command_input],142 outputs=[143 detection_image,144 detections_json,145 action_plan_md,146 action_image,147 status_output,148 ],149 )150 151 return demo152 