Alae65/anycoder
0
1import os2import re3from http import HTTPStatus4from typing import Dict, List, Optional, Tuple5import base646import mimetypes7import numpy as np8from PIL import Image9import requests10from urllib.parse import urlparse, urljoin11from bs4 import BeautifulSoup12import html2text13import json14import time15import webbrowser16import urllib.parse17import copy18import html19 20import gradio as gr21from huggingface_hub import InferenceClient22from huggingface_hub import HfApi23import tempfile24from openai import OpenAI25import uuid26import datetime27from mistralai import Mistral28import shutil29import urllib.parse30import mimetypes31import threading32import atexit33import asyncio34from datetime import datetime, timedelta35from typing import Optional36import dashscope37from dashscope.utils.oss_utils import check_and_upload_local38 39# Gradio supported languages for syntax highlighting40GRADIO_SUPPORTED_LANGUAGES = [41 "python", "json", "html"42]43 44def get_gradio_language(language):45 # Map composite options to a supported syntax highlighting46 if language == "streamlit":47 return "python"48 if language == "gradio":49 return "python"50 if language == "comfyui":51 return "json"52 return language if language in GRADIO_SUPPORTED_LANGUAGES else None53 54# Search/Replace Constants55SEARCH_START = "<<<<<<< SEARCH"56DIVIDER = "======="57REPLACE_END = ">>>>>>> REPLACE"58 59# Gradio Documentation Auto-Update System60GRADIO_LLMS_TXT_URL = "https://www.gradio.app/llms.txt"61GRADIO_DOCS_CACHE_FILE = ".gradio_docs_cache.txt"62GRADIO_DOCS_LAST_UPDATE_FILE = ".gradio_docs_last_update.txt"63GRADIO_DOCS_UPDATE_ON_APP_UPDATE = True # Only update when app is updated, not on a timer64 65# Global variable to store the current Gradio documentation66_gradio_docs_content: str | None = None67_gradio_docs_last_fetched: Optional[datetime] = None68 69# ComfyUI Documentation Auto-Update System70COMFYUI_LLMS_TXT_URL = "https://docs.comfy.org/llms.txt"71COMFYUI_DOCS_CACHE_FILE = ".comfyui_docs_cache.txt"72COMFYUI_DOCS_LAST_UPDATE_FILE = ".comfyui_docs_last_update.txt"73COMFYUI_DOCS_UPDATE_ON_APP_UPDATE = True # Only update when app is updated, not on a timer74 75# Global variable to store the current ComfyUI documentation76_comfyui_docs_content: str | None = None77_comfyui_docs_last_fetched: Optional[datetime] = None78 79# FastRTC Documentation Auto-Update System80FASTRTC_LLMS_TXT_URL = "https://fastrtc.org/llms.txt"81FASTRTC_DOCS_CACHE_FILE = ".fastrtc_docs_cache.txt"82FASTRTC_DOCS_LAST_UPDATE_FILE = ".fastrtc_docs_last_update.txt"83FASTRTC_DOCS_UPDATE_ON_APP_UPDATE = True # Only update when app is updated, not on a timer84 85# Global variable to store the current FastRTC documentation86_fastrtc_docs_content: str | None = None87_fastrtc_docs_last_fetched: Optional[datetime] = None88 89def fetch_gradio_docs() -> str | None:90 """Fetch the latest Gradio documentation from llms.txt"""91 try:92 response = requests.get(GRADIO_LLMS_TXT_URL, timeout=10)93 response.raise_for_status()94 return response.text95 except Exception as e:96 print(f"Warning: Failed to fetch Gradio docs from {GRADIO_LLMS_TXT_URL}: {e}")97 return None98 99def fetch_comfyui_docs() -> str | None:100 """Fetch the latest ComfyUI documentation from llms.txt"""101 try:102 response = requests.get(COMFYUI_LLMS_TXT_URL, timeout=10)103 response.raise_for_status()104 return response.text105 except Exception as e:106 print(f"Warning: Failed to fetch ComfyUI docs from {COMFYUI_LLMS_TXT_URL}: {e}")107 return None108 109def fetch_fastrtc_docs() -> str | None:110 """Fetch the latest FastRTC documentation from llms.txt"""111 try:112 response = requests.get(FASTRTC_LLMS_TXT_URL, timeout=10)113 response.raise_for_status()114 return response.text115 except Exception as e:116 print(f"Warning: Failed to fetch FastRTC docs from {FASTRTC_LLMS_TXT_URL}: {e}")117 return None118 119def filter_problematic_instructions(content: str) -> str:120 """Filter out problematic instructions that cause LLM to stop generation prematurely"""121 if not content:122 return content123 124 # List of problematic phrases that cause early termination when LLM encounters ``` in user code125 problematic_patterns = [126 r"Output ONLY the code inside a ``` code block, and do not include any explanations or extra text",127 r"output only the code inside a ```.*?``` code block",128 r"Always output only the.*?code.*?inside.*?```.*?```.*?block",129 r"Return ONLY the code inside a.*?```.*?``` code block",130 r"Do NOT add the language name at the top of the code output",131 r"do not include any explanations or extra text",132 r"Always output only the.*?code blocks.*?shown above, and do not include any explanations",133 r"Output.*?ONLY.*?code.*?inside.*?```.*?```",134 r"Return.*?ONLY.*?code.*?inside.*?```.*?```",135 r"Generate.*?ONLY.*?code.*?inside.*?```.*?```",136 r"Provide.*?ONLY.*?code.*?inside.*?```.*?```",137 ]138 139 # Remove problematic patterns140 filtered_content = content141 for pattern in problematic_patterns:142 # Use case-insensitive matching143 filtered_content = re.sub(pattern, "", filtered_content, flags=re.IGNORECASE | re.DOTALL)144 145 # Clean up any double newlines or extra whitespace left by removals146 filtered_content = re.sub(r'\n\s*\n\s*\n', '\n\n', filtered_content)147 filtered_content = re.sub(r'^\s+', '', filtered_content, flags=re.MULTILINE)148 149 return filtered_content150 151def load_cached_gradio_docs() -> str | None:152 """Load cached Gradio documentation from file"""153 try:154 if os.path.exists(GRADIO_DOCS_CACHE_FILE):155 with open(GRADIO_DOCS_CACHE_FILE, 'r', encoding='utf-8') as f:156 return f.read()157 except Exception as e:158 print(f"Warning: Failed to load cached Gradio docs: {e}")159 return None160 161def save_gradio_docs_cache(content: str):162 """Save Gradio documentation to cache file"""163 try:164 with open(GRADIO_DOCS_CACHE_FILE, 'w', encoding='utf-8') as f:165 f.write(content)166 with open(GRADIO_DOCS_LAST_UPDATE_FILE, 'w', encoding='utf-8') as f:167 f.write(datetime.now().isoformat())168 except Exception as e:169 print(f"Warning: Failed to save Gradio docs cache: {e}")170 171def load_comfyui_docs_cache() -> str | None:172 """Load ComfyUI documentation from cache file"""173 try:174 if os.path.exists(COMFYUI_DOCS_CACHE_FILE):175 with open(COMFYUI_DOCS_CACHE_FILE, 'r', encoding='utf-8') as f:176 return f.read()177 except Exception as e:178 print(f"Warning: Failed to load cached ComfyUI docs: {e}")179 return None180 181def save_comfyui_docs_cache(content: str):182 """Save ComfyUI documentation to cache file"""183 try:184 with open(COMFYUI_DOCS_CACHE_FILE, 'w', encoding='utf-8') as f:185 f.write(content)186 with open(COMFYUI_DOCS_LAST_UPDATE_FILE, 'w', encoding='utf-8') as f:187 f.write(datetime.now().isoformat())188 except Exception as e:189 print(f"Warning: Failed to save ComfyUI docs cache: {e}")190 191def load_fastrtc_docs_cache() -> str | None:192 """Load FastRTC documentation from cache file"""193 try:194 if os.path.exists(FASTRTC_DOCS_CACHE_FILE):195 with open(FASTRTC_DOCS_CACHE_FILE, 'r', encoding='utf-8') as f:196 return f.read()197 except Exception as e:198 print(f"Warning: Failed to load cached FastRTC docs: {e}")199 return None200 201def save_fastrtc_docs_cache(content: str):202 """Save FastRTC documentation to cache file"""203 try:204 with open(FASTRTC_DOCS_CACHE_FILE, 'w', encoding='utf-8') as f:205 f.write(content)206 with open(FASTRTC_DOCS_LAST_UPDATE_FILE, 'w', encoding='utf-8') as f:207 f.write(datetime.now().isoformat())208 except Exception as e:209 print(f"Warning: Failed to save FastRTC docs cache: {e}")210 211def get_last_update_time() -> Optional[datetime]:212 """Get the last update time from file"""213 try:214 if os.path.exists(GRADIO_DOCS_LAST_UPDATE_FILE):215 with open(GRADIO_DOCS_LAST_UPDATE_FILE, 'r', encoding='utf-8') as f:216 return datetime.fromisoformat(f.read().strip())217 except Exception as e:218 print(f"Warning: Failed to read last update time: {e}")219 return None220 221def should_update_gradio_docs() -> bool:222 """Check if Gradio documentation should be updated"""223 # Only update if we don't have cached content (first run or cache deleted)224 return not os.path.exists(GRADIO_DOCS_CACHE_FILE)225 226def should_update_comfyui_docs() -> bool:227 """Check if ComfyUI documentation should be updated"""228 # Only update if we don't have cached content (first run or cache deleted)229 return not os.path.exists(COMFYUI_DOCS_CACHE_FILE)230 231def should_update_fastrtc_docs() -> bool:232 """Check if FastRTC documentation should be updated"""233 # Only update if we don't have cached content (first run or cache deleted)234 return not os.path.exists(FASTRTC_DOCS_CACHE_FILE)235 236def force_update_gradio_docs():237 """238 Force an update of Gradio documentation (useful when app is updated).239 240 To manually refresh docs, you can call this function or simply delete the cache file:241 rm .gradio_docs_cache.txt && restart the app242 """243 global _gradio_docs_content, _gradio_docs_last_fetched244 245 print("๐ Forcing Gradio documentation update...")246 latest_content = fetch_gradio_docs()247 248 if latest_content:249 # Filter out problematic instructions that cause early termination250 filtered_content = filter_problematic_instructions(latest_content)251 _gradio_docs_content = filtered_content252 _gradio_docs_last_fetched = datetime.now()253 save_gradio_docs_cache(filtered_content)254 update_gradio_system_prompts()255 print("โ
Gradio documentation updated successfully")256 return True257 else:258 print("โ Failed to update Gradio documentation")259 return False260 261def force_update_comfyui_docs():262 """263 Force an update of ComfyUI documentation (useful when app is updated).264 265 To manually refresh docs, you can call this function or simply delete the cache file:266 rm .comfyui_docs_cache.txt && restart the app267 """268 global _comfyui_docs_content, _comfyui_docs_last_fetched269 270 print("๐ Forcing ComfyUI documentation update...")271 latest_content = fetch_comfyui_docs()272 273 if latest_content:274 # Filter out problematic instructions that cause early termination275 filtered_content = filter_problematic_instructions(latest_content)276 _comfyui_docs_content = filtered_content277 _comfyui_docs_last_fetched = datetime.now()278 save_comfyui_docs_cache(filtered_content)279 update_json_system_prompts()280 print("โ
ComfyUI documentation updated successfully")281 return True282 else:283 print("โ Failed to update ComfyUI documentation")284 return False285 286def force_update_fastrtc_docs():287 """288 Force an update of FastRTC documentation (useful when app is updated).289 290 To manually refresh docs, you can call this function or simply delete the cache file:291 rm .fastrtc_docs_cache.txt && restart the app292 """293 global _fastrtc_docs_content, _fastrtc_docs_last_fetched294 295 print("๐ Forcing FastRTC documentation update...")296 latest_content = fetch_fastrtc_docs()297 298 if latest_content:299 # Filter out problematic instructions that cause early termination300 filtered_content = filter_problematic_instructions(latest_content)301 _fastrtc_docs_content = filtered_content302 _fastrtc_docs_last_fetched = datetime.now()303 save_fastrtc_docs_cache(filtered_content)304 update_gradio_system_prompts()305 print("โ
FastRTC documentation updated successfully")306 return True307 else:308 print("โ Failed to update FastRTC documentation")309 return False310 311def get_gradio_docs_content() -> str:312 """Get the current Gradio documentation content, updating if necessary"""313 global _gradio_docs_content, _gradio_docs_last_fetched314 315 # Check if we need to update316 if (_gradio_docs_content is None or 317 _gradio_docs_last_fetched is None or 318 should_update_gradio_docs()):319 320 print("Updating Gradio documentation...")321 322 # Try to fetch latest content323 latest_content = fetch_gradio_docs()324 325 if latest_content:326 # Filter out problematic instructions that cause early termination327 filtered_content = filter_problematic_instructions(latest_content)328 _gradio_docs_content = filtered_content329 _gradio_docs_last_fetched = datetime.now()330 save_gradio_docs_cache(filtered_content)331 print("โ
Gradio documentation updated successfully")332 else:333 # Fallback to cached content334 cached_content = load_cached_gradio_docs()335 if cached_content:336 _gradio_docs_content = cached_content337 _gradio_docs_last_fetched = datetime.now()338 print("โ ๏ธ Using cached Gradio documentation (network fetch failed)")339 else:340 # Fallback to minimal content341 _gradio_docs_content = """342 # Gradio API Reference (Offline Fallback)343 344 This is a minimal fallback when documentation cannot be fetched.345 Please check your internet connection for the latest API reference.346 347 Basic Gradio components: Button, Textbox, Slider, Image, Audio, Video, File, etc.348 Use gr.Blocks() for custom layouts and gr.Interface() for simple apps.349 """350 print("โ Using minimal fallback documentation")351 352 return _gradio_docs_content or ""353 354def get_comfyui_docs_content() -> str:355 """Get the current ComfyUI documentation content, updating if necessary"""356 global _comfyui_docs_content, _comfyui_docs_last_fetched357 358 # Check if we need to update359 if (_comfyui_docs_content is None or 360 _comfyui_docs_last_fetched is None or 361 should_update_comfyui_docs()):362 363 print("Updating ComfyUI documentation...")364 365 # Try to fetch latest content366 latest_content = fetch_comfyui_docs()367 368 if latest_content:369 # Filter out problematic instructions that cause early termination370 filtered_content = filter_problematic_instructions(latest_content)371 _comfyui_docs_content = filtered_content372 _comfyui_docs_last_fetched = datetime.now()373 save_comfyui_docs_cache(filtered_content)374 print("โ
ComfyUI documentation updated successfully")375 else:376 # Fallback to cached content377 cached_content = load_comfyui_docs_cache()378 if cached_content:379 _comfyui_docs_content = cached_content380 _comfyui_docs_last_fetched = datetime.now()381 print("โ ๏ธ Using cached ComfyUI documentation (network fetch failed)")382 else:383 # Fallback to minimal content384 _comfyui_docs_content = """385 # ComfyUI API Reference (Offline Fallback)386 387 This is a minimal fallback when documentation cannot be fetched.388 Please check your internet connection for the latest API reference.389 390 Basic ComfyUI workflow structure: nodes, connections, inputs, outputs.391 Use CheckpointLoaderSimple, CLIPTextEncode, KSampler for basic workflows.392 """393 print("โ Using minimal fallback documentation")394 395 return _comfyui_docs_content or ""396 397def get_fastrtc_docs_content() -> str:398 """Get the current FastRTC documentation content, updating if necessary"""399 global _fastrtc_docs_content, _fastrtc_docs_last_fetched400 401 # Check if we need to update402 if (_fastrtc_docs_content is None or 403 _fastrtc_docs_last_fetched is None or 404 should_update_fastrtc_docs()):405 406 print("Updating FastRTC documentation...")407 408 # Try to fetch latest content409 latest_content = fetch_fastrtc_docs()410 411 if latest_content:412 # Filter out problematic instructions that cause early termination413 filtered_content = filter_problematic_instructions(latest_content)414 _fastrtc_docs_content = filtered_content415 _fastrtc_docs_last_fetched = datetime.now()416 save_fastrtc_docs_cache(filtered_content)417 print("โ
FastRTC documentation updated successfully")418 else:419 # Fallback to cached content420 cached_content = load_fastrtc_docs_cache()421 if cached_content:422 _fastrtc_docs_content = cached_content423 _fastrtc_docs_last_fetched = datetime.now()424 print("โ ๏ธ Using cached FastRTC documentation (network fetch failed)")425 else:426 # Fallback to minimal content427 _fastrtc_docs_content = """428 # FastRTC API Reference (Offline Fallback)429 430 This is a minimal fallback when documentation cannot be fetched.431 Please check your internet connection for the latest API reference.432 433 Basic FastRTC usage: Stream class, handlers, real-time audio/video processing.434 Use Stream(handler, modality, mode) for real-time communication apps.435 """436 print("โ Using minimal fallback documentation")437 438 return _fastrtc_docs_content or ""439 440def update_gradio_system_prompts():441 """Update the global Gradio system prompts with latest documentation"""442 global GRADIO_SYSTEM_PROMPT, GRADIO_SYSTEM_PROMPT_WITH_SEARCH443 444 docs_content = get_gradio_docs_content()445 fastrtc_content = get_fastrtc_docs_content()446 447 # Base system prompt448 base_prompt = """You are an expert Gradio developer. Create a complete, working Gradio application based on the user's request. Generate all necessary code to make the application functional and runnable.449 450## Multi-File Application Structure451 452When creating complex Gradio applications, organize your code into multiple files for better maintainability:453 454**File Organization:**455- `app.py` - Main application entry point with Gradio interface456- `utils.py` - Utility functions and helpers457- `models.py` - Model loading and inference functions458- `config.py` - Configuration and constants459- `requirements.txt` - Python dependencies460- Additional modules as needed (e.g., `data_processing.py`, `ui_components.py`)461 462**Output Format for Multi-File Apps:**463When generating multi-file applications, use this exact format:464 465```466=== app.py ===467[main application code]468 469=== utils.py ===470[utility functions]471 472=== requirements.txt ===473[dependencies]474```475 476**๐จ CRITICAL: Always Generate requirements.txt for New Applications**477- ALWAYS include requirements.txt when creating new Gradio applications478- Generate comprehensive, production-ready dependencies based on your code479- Include not just direct imports but also commonly needed companion packages480- Use correct PyPI package names (e.g., PIL โ Pillow, sklearn โ scikit-learn)481- For diffusers: use `git+https://github.com/huggingface/diffusers`482- For transformers: use `git+https://github.com/huggingface/transformers`483- Include supporting packages (accelerate, torch, tokenizers, etc.) when using ML libraries484- Your requirements.txt should ensure the application works smoothly in production485 486**Single vs Multi-File Decision:**487- Use single file for simple applications (< 100 lines) - but still generate requirements.txt if dependencies exist488- Use multi-file structure for complex applications with:489 - Multiple models or processing pipelines490 - Extensive utility functions491 - Complex UI with many components492 - Data processing workflows493 - When user specifically requests modular structure494 495๐จ IMPORTANT: If the user is asking to use external APIs (like OpenRouter, OpenAI API, Hugging Face Inference API, etc.), DO NOT use @spaces.GPU decorators or any ZeroGPU features. External APIs handle the model inference remotely, so GPU allocation on the Spaces instance is not needed.496 497๐จ CRITICAL REQUIREMENT: If the user provides ANY diffusion model code (FLUX, Stable Diffusion, etc.) that runs locally (not via API), you MUST implement ZeroGPU ahead-of-time (AoT) compilation. This is mandatory and provides 1.3x-1.8x performance improvements. Do not create basic Gradio apps without AoT optimization for diffusion models.498 499## ZeroGPU Integration (MANDATORY)500 501ALWAYS use ZeroGPU for GPU-dependent functions in Gradio apps:502 5031. Import the spaces module: `import spaces`5042. Decorate GPU-dependent functions with `@spaces.GPU`5053. Specify appropriate duration based on expected runtime:506 - Quick inference (< 30s): `@spaces.GPU(duration=30)`507 - Standard generation (30-60s): `@spaces.GPU` (default 60s)508 - Complex generation (60-120s): `@spaces.GPU(duration=120)`509 - Heavy processing (120-180s): `@spaces.GPU(duration=180)`510 511Example usage:512```python513import spaces514from diffusers import DiffusionPipeline515 516pipe = DiffusionPipeline.from_pretrained(...)517pipe.to('cuda')518 519@spaces.GPU(duration=120)520def generate(prompt):521 return pipe(prompt).images522 523gr.Interface(524 fn=generate,525 inputs=gr.Text(),526 outputs=gr.Gallery(),527).launch()528```529 530Duration Guidelines:531- Shorter durations improve queue priority for users532- Text-to-image: typically 30-60 seconds533- Image-to-image: typically 20-40 seconds 534- Video generation: typically 60-180 seconds535- Audio/music generation: typically 30-90 seconds536- Model loading + inference: add 10-30s buffer537- AoT compilation during startup: use @spaces.GPU(duration=1500) for maximum allowed duration538 539Functions that typically need @spaces.GPU:540- Image generation (text-to-image, image-to-image)541- Video generation542- Audio/music generation543- Model inference with transformers, diffusers544- Any function using .to('cuda') or GPU operations545 546## CRITICAL: Use ZeroGPU AoT Compilation for ALL Diffusion Models547 548FOR ANY DIFFUSION MODEL (FLUX, Stable Diffusion, etc.), YOU MUST IMPLEMENT AHEAD-OF-TIME COMPILATION.549This is NOT optional - it provides 1.3x-1.8x speedup and is essential for production ZeroGPU Spaces.550 551ALWAYS implement this pattern for diffusion models:552 553### MANDATORY: Basic AoT Compilation Pattern554YOU MUST USE THIS EXACT PATTERN for any diffusion model (FLUX, Stable Diffusion, etc.):555 5561. ALWAYS add AoT compilation function with @spaces.GPU(duration=1500)5572. ALWAYS use spaces.aoti_capture to capture inputs5583. ALWAYS use torch.export.export to export the transformer5594. ALWAYS use spaces.aoti_compile to compile5605. ALWAYS use spaces.aoti_apply to apply to pipeline561 562### Required AoT Implementation563```python564import spaces565import torch566from diffusers import DiffusionPipeline567 568MODEL_ID = 'black-forest-labs/FLUX.1-dev'569pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)570pipe.to('cuda')571 572@spaces.GPU(duration=1500) # Maximum duration allowed during startup573def compile_transformer():574 # 1. Capture example inputs575 with spaces.aoti_capture(pipe.transformer) as call:576 pipe("arbitrary example prompt")577 578 # 2. Export the model579 exported = torch.export.export(580 pipe.transformer,581 args=call.args,582 kwargs=call.kwargs,583 )584 585 # 3. Compile the exported model586 return spaces.aoti_compile(exported)587 588# 4. Apply compiled model to pipeline589compiled_transformer = compile_transformer()590spaces.aoti_apply(compiled_transformer, pipe.transformer)591 592@spaces.GPU593def generate(prompt):594 return pipe(prompt).images595```596 597### Advanced Optimizations598 599#### FP8 Quantization (Additional 1.2x speedup on H200)600```python601from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig602 603@spaces.GPU(duration=1500)604def compile_transformer_with_quantization():605 # Quantize before export for FP8 speedup606 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())607 608 with spaces.aoti_capture(pipe.transformer) as call:609 pipe("arbitrary example prompt")610 611 exported = torch.export.export(612 pipe.transformer,613 args=call.args,614 kwargs=call.kwargs,615 )616 return spaces.aoti_compile(exported)617```618 619#### Dynamic Shapes (Variable input sizes)620```python621from torch.utils._pytree import tree_map622 623@spaces.GPU(duration=1500)624def compile_transformer_dynamic():625 with spaces.aoti_capture(pipe.transformer) as call:626 pipe("arbitrary example prompt")627 628 # Define dynamic dimension ranges (model-dependent)629 transformer_hidden_dim = torch.export.Dim('hidden', min=4096, max=8212)630 631 # Map argument names to dynamic dimensions632 transformer_dynamic_shapes = {633 "hidden_states": {1: transformer_hidden_dim}, 634 "img_ids": {0: transformer_hidden_dim},635 }636 637 # Create dynamic shapes structure638 dynamic_shapes = tree_map(lambda v: None, call.kwargs)639 dynamic_shapes.update(transformer_dynamic_shapes)640 641 exported = torch.export.export(642 pipe.transformer,643 args=call.args,644 kwargs=call.kwargs,645 dynamic_shapes=dynamic_shapes,646 )647 return spaces.aoti_compile(exported)648```649 650#### Multi-Compile for Different Resolutions651```python652@spaces.GPU(duration=1500)653def compile_multiple_resolutions():654 compiled_models = {}655 resolutions = [(512, 512), (768, 768), (1024, 1024)]656 657 for width, height in resolutions:658 # Capture inputs for specific resolution659 with spaces.aoti_capture(pipe.transformer) as call:660 pipe(f"test prompt {width}x{height}", width=width, height=height)661 662 exported = torch.export.export(663 pipe.transformer,664 args=call.args,665 kwargs=call.kwargs,666 )667 compiled_models[f"{width}x{height}"] = spaces.aoti_compile(exported)668 669 return compiled_models670 671# Usage with resolution dispatch672compiled_models = compile_multiple_resolutions()673 674@spaces.GPU675def generate_with_resolution(prompt, width=1024, height=1024):676 resolution_key = f"{width}x{height}"677 if resolution_key in compiled_models:678 # Temporarily apply the right compiled model679 spaces.aoti_apply(compiled_models[resolution_key], pipe.transformer)680 return pipe(prompt, width=width, height=height).images681```682 683#### FlashAttention-3 Integration684```python685from kernels import get_kernel686 687# Load pre-built FA3 kernel compatible with H200688try:689 vllm_flash_attn3 = get_kernel("kernels-community/vllm-flash-attn3")690 print("โ
FlashAttention-3 kernel loaded successfully")691except Exception as e:692 print(f"โ ๏ธ FlashAttention-3 not available: {e}")693 694# Custom attention processor example695class FlashAttention3Processor:696 def __call__(self, attn, hidden_states, encoder_hidden_states=None, attention_mask=None):697 # Use FA3 kernel for attention computation698 return vllm_flash_attn3(hidden_states, encoder_hidden_states, attention_mask)699 700# Apply FA3 processor to model701if 'vllm_flash_attn3' in locals():702 for name, module in pipe.transformer.named_modules():703 if hasattr(module, 'processor'):704 module.processor = FlashAttention3Processor()705```706 707### Complete Optimized Example708```python709import spaces710import torch711from diffusers import DiffusionPipeline712from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig713 714MODEL_ID = 'black-forest-labs/FLUX.1-dev'715pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)716pipe.to('cuda')717 718@spaces.GPU(duration=1500)719def compile_optimized_transformer():720 # Apply FP8 quantization721 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())722 723 # Capture inputs724 with spaces.aoti_capture(pipe.transformer) as call:725 pipe("optimization test prompt")726 727 # Export and compile728 exported = torch.export.export(729 pipe.transformer,730 args=call.args,731 kwargs=call.kwargs,732 )733 return spaces.aoti_compile(exported)734 735# Compile during startup736compiled_transformer = compile_optimized_transformer()737spaces.aoti_apply(compiled_transformer, pipe.transformer)738 739@spaces.GPU740def generate(prompt):741 return pipe(prompt).images742```743 744**Expected Performance Gains:**745- Basic AoT: 1.3x-1.8x speedup746- + FP8 Quantization: Additional 1.2x speedup 747- + FlashAttention-3: Additional attention speedup748- Total potential: 2x-3x faster inference749 750**Hardware Requirements:**751- FP8 quantization requires CUDA compute capability โฅ 9.0 (H200 โ
)752- FlashAttention-3 works on H200 hardware via kernels library753- Dynamic shapes add flexibility for variable input sizes754 755## Complete Gradio API Reference756 757This reference is automatically synced from https://www.gradio.app/llms.txt to ensure accuracy.758 759"""760 761 # Search-enabled prompt762 search_prompt = """You are an expert Gradio developer with access to real-time web search. Create a complete, working Gradio application based on the user's request. When needed, use web search to find current best practices or verify latest Gradio features. Generate all necessary code to make the application functional and runnable.763 764## Multi-File Application Structure765 766When creating complex Gradio applications, organize your code into multiple files for better maintainability:767 768**File Organization:**769- `app.py` - Main application entry point with Gradio interface770- `utils.py` - Utility functions and helpers771- `models.py` - Model loading and inference functions772- `config.py` - Configuration and constants773- `requirements.txt` - Python dependencies774- Additional modules as needed (e.g., `data_processing.py`, `ui_components.py`)775 776**Output Format for Multi-File Apps:**777When generating multi-file applications, use this exact format:778 779```780=== app.py ===781[main application code]782 783=== utils.py ===784[utility functions]785 786=== requirements.txt ===787[dependencies]788```789 790**Single vs Multi-File Decision:**791- Use single file for simple applications (< 100 lines) - but still generate requirements.txt if dependencies exist792- Use multi-file structure for complex applications with:793 - Multiple models or processing pipelines794 - Extensive utility functions795 - Complex UI with many components796 - Data processing workflows797 - When user specifically requests modular structure798 799๐จ IMPORTANT: If the user is asking to use external APIs (like OpenRouter, OpenAI API, Hugging Face Inference API, etc.), DO NOT use @spaces.GPU decorators or any ZeroGPU features. External APIs handle the model inference remotely, so GPU allocation on the Spaces instance is not needed.800 801๐จ CRITICAL REQUIREMENT: If the user provides ANY diffusion model code (FLUX, Stable Diffusion, etc.) that runs locally (not via API), you MUST implement ZeroGPU ahead-of-time (AoT) compilation. This is mandatory and provides 1.3x-1.8x performance improvements. Do not create basic Gradio apps without AoT optimization for diffusion models.802 803## ZeroGPU Integration (MANDATORY)804 805ALWAYS use ZeroGPU for GPU-dependent functions in Gradio apps:806 8071. Import the spaces module: `import spaces`8082. Decorate GPU-dependent functions with `@spaces.GPU`8093. Specify appropriate duration based on expected runtime:810 - Quick inference (< 30s): `@spaces.GPU(duration=30)`811 - Standard generation (30-60s): `@spaces.GPU` (default 60s)812 - Complex generation (60-120s): `@spaces.GPU(duration=120)`813 - Heavy processing (120-180s): `@spaces.GPU(duration=180)`814 815Example usage:816```python817import spaces818from diffusers import DiffusionPipeline819 820pipe = DiffusionPipeline.from_pretrained(...)821pipe.to('cuda')822 823@spaces.GPU(duration=120)824def generate(prompt):825 return pipe(prompt).images826 827gr.Interface(828 fn=generate,829 inputs=gr.Text(),830 outputs=gr.Gallery(),831).launch()832```833 834Duration Guidelines:835- Shorter durations improve queue priority for users836- Text-to-image: typically 30-60 seconds837- Image-to-image: typically 20-40 seconds 838- Video generation: typically 60-180 seconds839- Audio/music generation: typically 30-90 seconds840- Model loading + inference: add 10-30s buffer841- AoT compilation during startup: use @spaces.GPU(duration=1500) for maximum allowed duration842 843Functions that typically need @spaces.GPU:844- Image generation (text-to-image, image-to-image)845- Video generation846- Audio/music generation847- Model inference with transformers, diffusers848- Any function using .to('cuda') or GPU operations849 850## CRITICAL: Use ZeroGPU AoT Compilation for ALL Diffusion Models851 852FOR ANY DIFFUSION MODEL (FLUX, Stable Diffusion, etc.), YOU MUST IMPLEMENT AHEAD-OF-TIME COMPILATION.853This is NOT optional - it provides 1.3x-1.8x speedup and is essential for production ZeroGPU Spaces.854 855ALWAYS implement this pattern for diffusion models:856 857### MANDATORY: Basic AoT Compilation Pattern858YOU MUST USE THIS EXACT PATTERN for any diffusion model (FLUX, Stable Diffusion, etc.):859 8601. ALWAYS add AoT compilation function with @spaces.GPU(duration=1500)8612. ALWAYS use spaces.aoti_capture to capture inputs8623. ALWAYS use torch.export.export to export the transformer8634. ALWAYS use spaces.aoti_compile to compile8645. ALWAYS use spaces.aoti_apply to apply to pipeline865 866### Required AoT Implementation867 868For production Spaces with heavy models, use ahead-of-time (AoT) compilation for 1.3x-1.8x speedups:869 870### Basic AoT Compilation871```python872import spaces873import torch874from diffusers import DiffusionPipeline875 876MODEL_ID = 'black-forest-labs/FLUX.1-dev'877pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)878pipe.to('cuda')879 880@spaces.GPU(duration=1500) # Maximum duration allowed during startup881def compile_transformer():882 # 1. Capture example inputs883 with spaces.aoti_capture(pipe.transformer) as call:884 pipe("arbitrary example prompt")885 886 # 2. Export the model887 exported = torch.export.export(888 pipe.transformer,889 args=call.args,890 kwargs=call.kwargs,891 )892 893 # 3. Compile the exported model894 return spaces.aoti_compile(exported)895 896# 4. Apply compiled model to pipeline897compiled_transformer = compile_transformer()898spaces.aoti_apply(compiled_transformer, pipe.transformer)899 900@spaces.GPU901def generate(prompt):902 return pipe(prompt).images903```904 905### Advanced Optimizations906 907#### FP8 Quantization (Additional 1.2x speedup on H200)908```python909from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig910 911@spaces.GPU(duration=1500)912def compile_transformer_with_quantization():913 # Quantize before export for FP8 speedup914 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())915 916 with spaces.aoti_capture(pipe.transformer) as call:917 pipe("arbitrary example prompt")918 919 exported = torch.export.export(920 pipe.transformer,921 args=call.args,922 kwargs=call.kwargs,923 )924 return spaces.aoti_compile(exported)925```926 927#### Dynamic Shapes (Variable input sizes)928```python929from torch.utils._pytree import tree_map930 931@spaces.GPU(duration=1500)932def compile_transformer_dynamic():933 with spaces.aoti_capture(pipe.transformer) as call:934 pipe("arbitrary example prompt")935 936 # Define dynamic dimension ranges (model-dependent)937 transformer_hidden_dim = torch.export.Dim('hidden', min=4096, max=8212)938 939 # Map argument names to dynamic dimensions940 transformer_dynamic_shapes = {941 "hidden_states": {1: transformer_hidden_dim}, 942 "img_ids": {0: transformer_hidden_dim},943 }944 945 # Create dynamic shapes structure946 dynamic_shapes = tree_map(lambda v: None, call.kwargs)947 dynamic_shapes.update(transformer_dynamic_shapes)948 949 exported = torch.export.export(950 pipe.transformer,951 args=call.args,952 kwargs=call.kwargs,953 dynamic_shapes=dynamic_shapes,954 )955 return spaces.aoti_compile(exported)956```957 958#### Multi-Compile for Different Resolutions959```python960@spaces.GPU(duration=1500)961def compile_multiple_resolutions():962 compiled_models = {}963 resolutions = [(512, 512), (768, 768), (1024, 1024)]964 965 for width, height in resolutions:966 # Capture inputs for specific resolution967 with spaces.aoti_capture(pipe.transformer) as call:968 pipe(f"test prompt {width}x{height}", width=width, height=height)969 970 exported = torch.export.export(971 pipe.transformer,972 args=call.args,973 kwargs=call.kwargs,974 )975 compiled_models[f"{width}x{height}"] = spaces.aoti_compile(exported)976 977 return compiled_models978 979# Usage with resolution dispatch980compiled_models = compile_multiple_resolutions()981 982@spaces.GPU983def generate_with_resolution(prompt, width=1024, height=1024):984 resolution_key = f"{width}x{height}"985 if resolution_key in compiled_models:986 # Temporarily apply the right compiled model987 spaces.aoti_apply(compiled_models[resolution_key], pipe.transformer)988 return pipe(prompt, width=width, height=height).images989```990 991#### FlashAttention-3 Integration992```python993from kernels import get_kernel994 995# Load pre-built FA3 kernel compatible with H200996try:997 vllm_flash_attn3 = get_kernel("kernels-community/vllm-flash-attn3")998 print("โ
FlashAttention-3 kernel loaded successfully")999except Exception as e:1000 print(f"โ ๏ธ FlashAttention-3 not available: {e}")1001 1002# Custom attention processor example1003class FlashAttention3Processor:1004 def __call__(self, attn, hidden_states, encoder_hidden_states=None, attention_mask=None):1005 # Use FA3 kernel for attention computation1006 return vllm_flash_attn3(hidden_states, encoder_hidden_states, attention_mask)1007 1008# Apply FA3 processor to model1009if 'vllm_flash_attn3' in locals():1010 for name, module in pipe.transformer.named_modules():1011 if hasattr(module, 'processor'):1012 module.processor = FlashAttention3Processor()1013```1014 1015### Complete Optimized Example1016```python1017import spaces1018import torch1019from diffusers import DiffusionPipeline1020from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig1021 1022MODEL_ID = 'black-forest-labs/FLUX.1-dev'1023pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)1024pipe.to('cuda')1025 1026@spaces.GPU(duration=1500)1027def compile_optimized_transformer():1028 # Apply FP8 quantization1029 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())1030 1031 # Capture inputs1032 with spaces.aoti_capture(pipe.transformer) as call:1033 pipe("optimization test prompt")1034 1035 # Export and compile1036 exported = torch.export.export(1037 pipe.transformer,1038 args=call.args,1039 kwargs=call.kwargs,1040 )1041 return spaces.aoti_compile(exported)1042 1043# Compile during startup1044compiled_transformer = compile_optimized_transformer()1045spaces.aoti_apply(compiled_transformer, pipe.transformer)1046 1047@spaces.GPU1048def generate(prompt):1049 return pipe(prompt).images1050```1051 1052**Expected Performance Gains:**1053- Basic AoT: 1.3x-1.8x speedup1054- + FP8 Quantization: Additional 1.2x speedup 1055- + FlashAttention-3: Additional attention speedup1056- Total potential: 2x-3x faster inference1057 1058**Hardware Requirements:**1059- FP8 quantization requires CUDA compute capability โฅ 9.0 (H200 โ
)1060- FlashAttention-3 works on H200 hardware via kernels library1061- Dynamic shapes add flexibility for variable input sizes1062 1063## Complete Gradio API Reference1064 1065This reference is automatically synced from https://www.gradio.app/llms.txt to ensure accuracy.1066 1067"""1068 1069 # Add FastRTC documentation if available1070 if fastrtc_content.strip():1071 fastrtc_section = f"""1072## FastRTC Reference Documentation1073 1074When building real-time audio/video applications with Gradio, use this FastRTC reference:1075 1076{fastrtc_content}1077 1078This reference is automatically synced from https://fastrtc.org/llms.txt to ensure accuracy.1079 1080"""1081 base_prompt += fastrtc_section1082 search_prompt += fastrtc_section1083 1084 # Update the prompts1085 GRADIO_SYSTEM_PROMPT = base_prompt + docs_content + "\n\nAlways use the exact function signatures from this API reference and follow modern Gradio patterns.\n\nIMPORTANT: Always include \"Built with anycoder\" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder"1086 GRADIO_SYSTEM_PROMPT_WITH_SEARCH = search_prompt + docs_content + "\n\nAlways use the exact function signatures from this API reference and follow modern Gradio patterns.\n\nIMPORTANT: Always include \"Built with anycoder\" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder"1087 1088def update_json_system_prompts():1089 """Update the global JSON system prompts with latest ComfyUI documentation"""1090 global JSON_SYSTEM_PROMPT, JSON_SYSTEM_PROMPT_WITH_SEARCH1091 1092 docs_content = get_comfyui_docs_content()1093 1094 # Base system prompt for regular JSON1095 base_prompt = """You are an expert JSON developer. Generate clean, valid JSON data based on the user's request. Follow JSON syntax rules strictly:1096- Use double quotes for strings1097- No trailing commas1098- Proper nesting and structure1099- Valid data types (string, number, boolean, null, object, array)1100 1101Generate ONLY the JSON data requested - no HTML, no applications, no explanations outside the JSON. The output should be pure, valid JSON that can be parsed directly.1102 1103"""1104 1105 # Search-enabled system prompt for regular JSON1106 search_prompt = """You are an expert JSON developer. You have access to real-time web search. When needed, use web search to find the latest information or data structures for your JSON generation.1107 1108Generate clean, valid JSON data based on the user's request. Follow JSON syntax rules strictly:1109- Use double quotes for strings1110- No trailing commas1111- Proper nesting and structure1112- Valid data types (string, number, boolean, null, object, array)1113 1114Generate ONLY the JSON data requested - no HTML, no applications, no explanations outside the JSON. The output should be pure, valid JSON that can be parsed directly.1115 1116"""1117 1118 # Add ComfyUI documentation if available1119 if docs_content.strip():1120 comfyui_section = f"""1121## ComfyUI Reference Documentation1122 1123When generating JSON data related to ComfyUI workflows, nodes, or configurations, use this reference:1124 1125{docs_content}1126 1127This reference is automatically synced from https://docs.comfy.org/llms.txt to ensure accuracy.1128 1129"""1130 base_prompt += comfyui_section1131 search_prompt += comfyui_section1132 1133 # Update the prompts1134 JSON_SYSTEM_PROMPT = base_prompt1135 JSON_SYSTEM_PROMPT_WITH_SEARCH = search_prompt1136 1137def get_comfyui_system_prompt():1138 """Get ComfyUI-specific system prompt with enhanced guidance"""1139 docs_content = get_comfyui_docs_content()1140 1141 base_prompt = """You are an expert ComfyUI developer. Generate clean, valid JSON workflows for ComfyUI based on the user's request. 1142 1143ComfyUI workflows are JSON structures that define:1144- Nodes: Individual processing units with specific functions1145- Connections: Links between nodes that define data flow1146- Parameters: Configuration values for each node1147- Inputs/Outputs: Data flow between nodes1148 1149Follow JSON syntax rules strictly:1150- Use double quotes for strings1151- No trailing commas1152- Proper nesting and structure1153- Valid data types (string, number, boolean, null, object, array)1154 1155Generate ONLY the ComfyUI workflow JSON - no HTML, no applications, no explanations outside the JSON. The output should be a complete, valid ComfyUI workflow that can be loaded directly into ComfyUI.1156 1157"""1158 1159 # Add ComfyUI documentation if available1160 if docs_content.strip():1161 comfyui_section = f"""1162## ComfyUI Reference Documentation1163 1164Use this reference for accurate node types, parameters, and workflow structures:1165 1166{docs_content}1167 1168This reference is automatically synced from https://docs.comfy.org/llms.txt to ensure accuracy.1169 1170"""1171 base_prompt += comfyui_section1172 1173 base_prompt += """1174IMPORTANT: Always include "Built with anycoder" as a comment or metadata field in your ComfyUI workflow JSON that references https://huggingface.co/spaces/akhaliq/anycoder1175"""1176 1177 return base_prompt1178 1179# Initialize Gradio documentation on startup1180def initialize_gradio_docs():1181 """Initialize Gradio documentation on application startup"""1182 try:1183 update_gradio_system_prompts()1184 if should_update_gradio_docs():1185 print("๐ Gradio documentation system initialized (fetched fresh content)")1186 else:1187 print("๐ Gradio documentation system initialized (using cached content)")1188 except Exception as e:1189 print(f"Warning: Failed to initialize Gradio documentation: {e}")1190 1191# Initialize ComfyUI documentation on startup1192def initialize_comfyui_docs():1193 """Initialize ComfyUI documentation on application startup"""1194 try:1195 update_json_system_prompts()1196 if should_update_comfyui_docs():1197 print("๐ ComfyUI documentation system initialized (fetched fresh content)")1198 else:1199 print("๐ ComfyUI documentation system initialized (using cached content)")1200 except Exception as e: