cafe3310/anycoder
0
1import os2import re3from http import HTTPStatus4from typing import Dict, List, Optional, Tuple5import base646import mimetypes7import PyPDF28import docx9import cv210import numpy as np11from PIL import Image12import pytesseract13import requests14from urllib.parse import urlparse, urljoin15from bs4 import BeautifulSoup16import html2text17import json18import time19import webbrowser20import urllib.parse21import copy22import html23 24import gradio as gr25from huggingface_hub import InferenceClient26from tavily import TavilyClient27from huggingface_hub import HfApi28import tempfile29from openai import OpenAI30import uuid31import datetime32from mistralai import Mistral33import shutil34import urllib.parse35import mimetypes36import threading37import atexit38import asyncio39from datetime import datetime, timedelta40from typing import Optional41import dashscope42from dashscope.utils.oss_utils import check_and_upload_local43 44# Gradio supported languages for syntax highlighting45GRADIO_SUPPORTED_LANGUAGES = [46 "python", "c", "cpp", "markdown", "latex", "json", "html", "css", "javascript", "jinja2", "typescript", "yaml", "dockerfile", "shell", "r", "sql", "sql-msSQL", "sql-mySQL", "sql-mariaDB", "sql-sqlite", "sql-cassandra", "sql-plSQL", "sql-hive", "sql-pgSQL", "sql-gql", "sql-gpSQL", "sql-sparkSQL", "sql-esper", None47]48 49def get_gradio_language(language):50 # Map composite options to a supported syntax highlighting51 if language == "streamlit":52 return "python"53 if language == "gradio":54 return "python"55 return language if language in GRADIO_SUPPORTED_LANGUAGES else None56 57# Search/Replace Constants58SEARCH_START = "<<<<<<< SEARCH"59DIVIDER = "======="60REPLACE_END = ">>>>>>> REPLACE"61 62# Gradio Documentation Auto-Update System63GRADIO_LLMS_TXT_URL = "https://www.gradio.app/llms.txt"64GRADIO_DOCS_CACHE_FILE = ".gradio_docs_cache.txt"65GRADIO_DOCS_LAST_UPDATE_FILE = ".gradio_docs_last_update.txt"66GRADIO_DOCS_UPDATE_ON_APP_UPDATE = True # Only update when app is updated, not on a timer67 68# Global variable to store the current Gradio documentation69_gradio_docs_content: str | None = None70_gradio_docs_last_fetched: Optional[datetime] = None71 72# ComfyUI Documentation Auto-Update System73COMFYUI_LLMS_TXT_URL = "https://docs.comfy.org/llms.txt"74COMFYUI_DOCS_CACHE_FILE = ".comfyui_docs_cache.txt"75COMFYUI_DOCS_LAST_UPDATE_FILE = ".comfyui_docs_last_update.txt"76COMFYUI_DOCS_UPDATE_ON_APP_UPDATE = True # Only update when app is updated, not on a timer77 78# Global variable to store the current ComfyUI documentation79_comfyui_docs_content: str | None = None80_comfyui_docs_last_fetched: Optional[datetime] = None81 82def fetch_gradio_docs() -> str | None:83 """Fetch the latest Gradio documentation from llms.txt"""84 try:85 response = requests.get(GRADIO_LLMS_TXT_URL, timeout=10)86 response.raise_for_status()87 return response.text88 except Exception as e:89 print(f"Warning: Failed to fetch Gradio docs from {GRADIO_LLMS_TXT_URL}: {e}")90 return None91 92def fetch_comfyui_docs() -> str | None:93 """Fetch the latest ComfyUI documentation from llms.txt"""94 try:95 response = requests.get(COMFYUI_LLMS_TXT_URL, timeout=10)96 response.raise_for_status()97 return response.text98 except Exception as e:99 print(f"Warning: Failed to fetch ComfyUI docs from {COMFYUI_LLMS_TXT_URL}: {e}")100 return None101 102def filter_problematic_instructions(content: str) -> str:103 """Filter out problematic instructions that cause LLM to stop generation prematurely"""104 if not content:105 return content106 107 # List of problematic phrases that cause early termination when LLM encounters ``` in user code108 problematic_patterns = [109 r"Output ONLY the code inside a ``` code block, and do not include any explanations or extra text",110 r"output only the code inside a ```.*?``` code block",111 r"Always output only the.*?code.*?inside.*?```.*?```.*?block",112 r"Return ONLY the code inside a.*?```.*?``` code block",113 r"Do NOT add the language name at the top of the code output",114 r"do not include any explanations or extra text",115 r"Always output only the.*?code blocks.*?shown above, and do not include any explanations",116 r"Output.*?ONLY.*?code.*?inside.*?```.*?```",117 r"Return.*?ONLY.*?code.*?inside.*?```.*?```",118 r"Generate.*?ONLY.*?code.*?inside.*?```.*?```",119 r"Provide.*?ONLY.*?code.*?inside.*?```.*?```",120 ]121 122 # Remove problematic patterns123 filtered_content = content124 for pattern in problematic_patterns:125 # Use case-insensitive matching126 filtered_content = re.sub(pattern, "", filtered_content, flags=re.IGNORECASE | re.DOTALL)127 128 # Clean up any double newlines or extra whitespace left by removals129 filtered_content = re.sub(r'\n\s*\n\s*\n', '\n\n', filtered_content)130 filtered_content = re.sub(r'^\s+', '', filtered_content, flags=re.MULTILINE)131 132 return filtered_content133 134def load_cached_gradio_docs() -> str | None:135 """Load cached Gradio documentation from file"""136 try:137 if os.path.exists(GRADIO_DOCS_CACHE_FILE):138 with open(GRADIO_DOCS_CACHE_FILE, 'r', encoding='utf-8') as f:139 return f.read()140 except Exception as e:141 print(f"Warning: Failed to load cached Gradio docs: {e}")142 return None143 144def save_gradio_docs_cache(content: str):145 """Save Gradio documentation to cache file"""146 try:147 with open(GRADIO_DOCS_CACHE_FILE, 'w', encoding='utf-8') as f:148 f.write(content)149 with open(GRADIO_DOCS_LAST_UPDATE_FILE, 'w', encoding='utf-8') as f:150 f.write(datetime.now().isoformat())151 except Exception as e:152 print(f"Warning: Failed to save Gradio docs cache: {e}")153 154def load_comfyui_docs_cache() -> str | None:155 """Load ComfyUI documentation from cache file"""156 try:157 if os.path.exists(COMFYUI_DOCS_CACHE_FILE):158 with open(COMFYUI_DOCS_CACHE_FILE, 'r', encoding='utf-8') as f:159 return f.read()160 except Exception as e:161 print(f"Warning: Failed to load cached ComfyUI docs: {e}")162 return None163 164def save_comfyui_docs_cache(content: str):165 """Save ComfyUI documentation to cache file"""166 try:167 with open(COMFYUI_DOCS_CACHE_FILE, 'w', encoding='utf-8') as f:168 f.write(content)169 with open(COMFYUI_DOCS_LAST_UPDATE_FILE, 'w', encoding='utf-8') as f:170 f.write(datetime.now().isoformat())171 except Exception as e:172 print(f"Warning: Failed to save ComfyUI docs cache: {e}")173 174def get_last_update_time() -> Optional[datetime]:175 """Get the last update time from file"""176 try:177 if os.path.exists(GRADIO_DOCS_LAST_UPDATE_FILE):178 with open(GRADIO_DOCS_LAST_UPDATE_FILE, 'r', encoding='utf-8') as f:179 return datetime.fromisoformat(f.read().strip())180 except Exception as e:181 print(f"Warning: Failed to read last update time: {e}")182 return None183 184def should_update_gradio_docs() -> bool:185 """Check if Gradio documentation should be updated"""186 # Only update if we don't have cached content (first run or cache deleted)187 return not os.path.exists(GRADIO_DOCS_CACHE_FILE)188 189def should_update_comfyui_docs() -> bool:190 """Check if ComfyUI documentation should be updated"""191 # Only update if we don't have cached content (first run or cache deleted)192 return not os.path.exists(COMFYUI_DOCS_CACHE_FILE)193 194def force_update_gradio_docs():195 """196 Force an update of Gradio documentation (useful when app is updated).197 198 To manually refresh docs, you can call this function or simply delete the cache file:199 rm .gradio_docs_cache.txt && restart the app200 """201 global _gradio_docs_content, _gradio_docs_last_fetched202 203 print("๐ Forcing Gradio documentation update...")204 latest_content = fetch_gradio_docs()205 206 if latest_content:207 # Filter out problematic instructions that cause early termination208 filtered_content = filter_problematic_instructions(latest_content)209 _gradio_docs_content = filtered_content210 _gradio_docs_last_fetched = datetime.now()211 save_gradio_docs_cache(filtered_content)212 update_gradio_system_prompts()213 print("โ
Gradio documentation updated successfully")214 return True215 else:216 print("โ Failed to update Gradio documentation")217 return False218 219def force_update_comfyui_docs():220 """221 Force an update of ComfyUI documentation (useful when app is updated).222 223 To manually refresh docs, you can call this function or simply delete the cache file:224 rm .comfyui_docs_cache.txt && restart the app225 """226 global _comfyui_docs_content, _comfyui_docs_last_fetched227 228 print("๐ Forcing ComfyUI documentation update...")229 latest_content = fetch_comfyui_docs()230 231 if latest_content:232 # Filter out problematic instructions that cause early termination233 filtered_content = filter_problematic_instructions(latest_content)234 _comfyui_docs_content = filtered_content235 _comfyui_docs_last_fetched = datetime.now()236 save_comfyui_docs_cache(filtered_content)237 update_json_system_prompts()238 print("โ
ComfyUI documentation updated successfully")239 return True240 else:241 print("โ Failed to update ComfyUI documentation")242 return False243 244def get_gradio_docs_content() -> str:245 """Get the current Gradio documentation content, updating if necessary"""246 global _gradio_docs_content, _gradio_docs_last_fetched247 248 # Check if we need to update249 if (_gradio_docs_content is None or 250 _gradio_docs_last_fetched is None or 251 should_update_gradio_docs()):252 253 print("Updating Gradio documentation...")254 255 # Try to fetch latest content256 latest_content = fetch_gradio_docs()257 258 if latest_content:259 # Filter out problematic instructions that cause early termination260 filtered_content = filter_problematic_instructions(latest_content)261 _gradio_docs_content = filtered_content262 _gradio_docs_last_fetched = datetime.now()263 save_gradio_docs_cache(filtered_content)264 print("โ
Gradio documentation updated successfully")265 else:266 # Fallback to cached content267 cached_content = load_cached_gradio_docs()268 if cached_content:269 _gradio_docs_content = cached_content270 _gradio_docs_last_fetched = datetime.now()271 print("โ ๏ธ Using cached Gradio documentation (network fetch failed)")272 else:273 # Fallback to minimal content274 _gradio_docs_content = """275 # Gradio API Reference (Offline Fallback)276 277 This is a minimal fallback when documentation cannot be fetched.278 Please check your internet connection for the latest API reference.279 280 Basic Gradio components: Button, Textbox, Slider, Image, Audio, Video, File, etc.281 Use gr.Blocks() for custom layouts and gr.Interface() for simple apps.282 """283 print("โ Using minimal fallback documentation")284 285 return _gradio_docs_content or ""286 287def get_comfyui_docs_content() -> str:288 """Get the current ComfyUI documentation content, updating if necessary"""289 global _comfyui_docs_content, _comfyui_docs_last_fetched290 291 # Check if we need to update292 if (_comfyui_docs_content is None or 293 _comfyui_docs_last_fetched is None or 294 should_update_comfyui_docs()):295 296 print("Updating ComfyUI documentation...")297 298 # Try to fetch latest content299 latest_content = fetch_comfyui_docs()300 301 if latest_content:302 # Filter out problematic instructions that cause early termination303 filtered_content = filter_problematic_instructions(latest_content)304 _comfyui_docs_content = filtered_content305 _comfyui_docs_last_fetched = datetime.now()306 save_comfyui_docs_cache(filtered_content)307 print("โ
ComfyUI documentation updated successfully")308 else:309 # Fallback to cached content310 cached_content = load_comfyui_docs_cache()311 if cached_content:312 _comfyui_docs_content = cached_content313 _comfyui_docs_last_fetched = datetime.now()314 print("โ ๏ธ Using cached ComfyUI documentation (network fetch failed)")315 else:316 # Fallback to minimal content317 _comfyui_docs_content = """318 # ComfyUI API Reference (Offline Fallback)319 320 This is a minimal fallback when documentation cannot be fetched.321 Please check your internet connection for the latest API reference.322 323 Basic ComfyUI workflow structure: nodes, connections, inputs, outputs.324 Use CheckpointLoaderSimple, CLIPTextEncode, KSampler for basic workflows.325 """326 print("โ Using minimal fallback documentation")327 328 return _comfyui_docs_content or ""329 330def update_gradio_system_prompts():331 """Update the global Gradio system prompts with latest documentation"""332 global GRADIO_SYSTEM_PROMPT, GRADIO_SYSTEM_PROMPT_WITH_SEARCH333 334 docs_content = get_gradio_docs_content()335 336 # Base system prompt337 base_prompt = """You are an expert Gradio developer. Create a complete, working Gradio application based on the user's request. Generate all necessary code to make the application functional and runnable.338 339๐จ IMPORTANT: If the user is asking to use external APIs (like OpenRouter, OpenAI API, Hugging Face Inference API, etc.), DO NOT use @spaces.GPU decorators or any ZeroGPU features. External APIs handle the model inference remotely, so GPU allocation on the Spaces instance is not needed.340 341๐จ CRITICAL REQUIREMENT: If the user provides ANY diffusion model code (FLUX, Stable Diffusion, etc.) that runs locally (not via API), you MUST implement ZeroGPU ahead-of-time (AoT) compilation. This is mandatory and provides 1.3x-1.8x performance improvements. Do not create basic Gradio apps without AoT optimization for diffusion models.342 343## ZeroGPU Integration (MANDATORY)344 345ALWAYS use ZeroGPU for GPU-dependent functions in Gradio apps:346 3471. Import the spaces module: `import spaces`3482. Decorate GPU-dependent functions with `@spaces.GPU`3493. Specify appropriate duration based on expected runtime:350 - Quick inference (< 30s): `@spaces.GPU(duration=30)`351 - Standard generation (30-60s): `@spaces.GPU` (default 60s)352 - Complex generation (60-120s): `@spaces.GPU(duration=120)`353 - Heavy processing (120-180s): `@spaces.GPU(duration=180)`354 355Example usage:356```python357import spaces358from diffusers import DiffusionPipeline359 360pipe = DiffusionPipeline.from_pretrained(...)361pipe.to('cuda')362 363@spaces.GPU(duration=120)364def generate(prompt):365 return pipe(prompt).images366 367gr.Interface(368 fn=generate,369 inputs=gr.Text(),370 outputs=gr.Gallery(),371).launch()372```373 374Duration Guidelines:375- Shorter durations improve queue priority for users376- Text-to-image: typically 30-60 seconds377- Image-to-image: typically 20-40 seconds 378- Video generation: typically 60-180 seconds379- Audio/music generation: typically 30-90 seconds380- Model loading + inference: add 10-30s buffer381- AoT compilation during startup: use @spaces.GPU(duration=1500) for maximum allowed duration382 383Functions that typically need @spaces.GPU:384- Image generation (text-to-image, image-to-image)385- Video generation386- Audio/music generation387- Model inference with transformers, diffusers388- Any function using .to('cuda') or GPU operations389 390## CRITICAL: Use ZeroGPU AoT Compilation for ALL Diffusion Models391 392FOR ANY DIFFUSION MODEL (FLUX, Stable Diffusion, etc.), YOU MUST IMPLEMENT AHEAD-OF-TIME COMPILATION.393This is NOT optional - it provides 1.3x-1.8x speedup and is essential for production ZeroGPU Spaces.394 395ALWAYS implement this pattern for diffusion models:396 397### MANDATORY: Basic AoT Compilation Pattern398YOU MUST USE THIS EXACT PATTERN for any diffusion model (FLUX, Stable Diffusion, etc.):399 4001. ALWAYS add AoT compilation function with @spaces.GPU(duration=1500)4012. ALWAYS use spaces.aoti_capture to capture inputs4023. ALWAYS use torch.export.export to export the transformer4034. ALWAYS use spaces.aoti_compile to compile4045. ALWAYS use spaces.aoti_apply to apply to pipeline405 406### Required AoT Implementation407```python408import spaces409import torch410from diffusers import DiffusionPipeline411 412MODEL_ID = 'black-forest-labs/FLUX.1-dev'413pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)414pipe.to('cuda')415 416@spaces.GPU(duration=1500) # Maximum duration allowed during startup417def compile_transformer():418 # 1. Capture example inputs419 with spaces.aoti_capture(pipe.transformer) as call:420 pipe("arbitrary example prompt")421 422 # 2. Export the model423 exported = torch.export.export(424 pipe.transformer,425 args=call.args,426 kwargs=call.kwargs,427 )428 429 # 3. Compile the exported model430 return spaces.aoti_compile(exported)431 432# 4. Apply compiled model to pipeline433compiled_transformer = compile_transformer()434spaces.aoti_apply(compiled_transformer, pipe.transformer)435 436@spaces.GPU437def generate(prompt):438 return pipe(prompt).images439```440 441### Advanced Optimizations442 443#### FP8 Quantization (Additional 1.2x speedup on H200)444```python445from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig446 447@spaces.GPU(duration=1500)448def compile_transformer_with_quantization():449 # Quantize before export for FP8 speedup450 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())451 452 with spaces.aoti_capture(pipe.transformer) as call:453 pipe("arbitrary example prompt")454 455 exported = torch.export.export(456 pipe.transformer,457 args=call.args,458 kwargs=call.kwargs,459 )460 return spaces.aoti_compile(exported)461```462 463#### Dynamic Shapes (Variable input sizes)464```python465from torch.utils._pytree import tree_map466 467@spaces.GPU(duration=1500)468def compile_transformer_dynamic():469 with spaces.aoti_capture(pipe.transformer) as call:470 pipe("arbitrary example prompt")471 472 # Define dynamic dimension ranges (model-dependent)473 transformer_hidden_dim = torch.export.Dim('hidden', min=4096, max=8212)474 475 # Map argument names to dynamic dimensions476 transformer_dynamic_shapes = {477 "hidden_states": {1: transformer_hidden_dim}, 478 "img_ids": {0: transformer_hidden_dim},479 }480 481 # Create dynamic shapes structure482 dynamic_shapes = tree_map(lambda v: None, call.kwargs)483 dynamic_shapes.update(transformer_dynamic_shapes)484 485 exported = torch.export.export(486 pipe.transformer,487 args=call.args,488 kwargs=call.kwargs,489 dynamic_shapes=dynamic_shapes,490 )491 return spaces.aoti_compile(exported)492```493 494#### Multi-Compile for Different Resolutions495```python496@spaces.GPU(duration=1500)497def compile_multiple_resolutions():498 compiled_models = {}499 resolutions = [(512, 512), (768, 768), (1024, 1024)]500 501 for width, height in resolutions:502 # Capture inputs for specific resolution503 with spaces.aoti_capture(pipe.transformer) as call:504 pipe(f"test prompt {width}x{height}", width=width, height=height)505 506 exported = torch.export.export(507 pipe.transformer,508 args=call.args,509 kwargs=call.kwargs,510 )511 compiled_models[f"{width}x{height}"] = spaces.aoti_compile(exported)512 513 return compiled_models514 515# Usage with resolution dispatch516compiled_models = compile_multiple_resolutions()517 518@spaces.GPU519def generate_with_resolution(prompt, width=1024, height=1024):520 resolution_key = f"{width}x{height}"521 if resolution_key in compiled_models:522 # Temporarily apply the right compiled model523 spaces.aoti_apply(compiled_models[resolution_key], pipe.transformer)524 return pipe(prompt, width=width, height=height).images525```526 527#### FlashAttention-3 Integration528```python529from kernels import get_kernel530 531# Load pre-built FA3 kernel compatible with H200532try:533 vllm_flash_attn3 = get_kernel("kernels-community/vllm-flash-attn3")534 print("โ
FlashAttention-3 kernel loaded successfully")535except Exception as e:536 print(f"โ ๏ธ FlashAttention-3 not available: {e}")537 538# Custom attention processor example539class FlashAttention3Processor:540 def __call__(self, attn, hidden_states, encoder_hidden_states=None, attention_mask=None):541 # Use FA3 kernel for attention computation542 return vllm_flash_attn3(hidden_states, encoder_hidden_states, attention_mask)543 544# Apply FA3 processor to model545if 'vllm_flash_attn3' in locals():546 for name, module in pipe.transformer.named_modules():547 if hasattr(module, 'processor'):548 module.processor = FlashAttention3Processor()549```550 551### Complete Optimized Example552```python553import spaces554import torch555from diffusers import DiffusionPipeline556from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig557 558MODEL_ID = 'black-forest-labs/FLUX.1-dev'559pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)560pipe.to('cuda')561 562@spaces.GPU(duration=1500)563def compile_optimized_transformer():564 # Apply FP8 quantization565 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())566 567 # Capture inputs568 with spaces.aoti_capture(pipe.transformer) as call:569 pipe("optimization test prompt")570 571 # Export and compile572 exported = torch.export.export(573 pipe.transformer,574 args=call.args,575 kwargs=call.kwargs,576 )577 return spaces.aoti_compile(exported)578 579# Compile during startup580compiled_transformer = compile_optimized_transformer()581spaces.aoti_apply(compiled_transformer, pipe.transformer)582 583@spaces.GPU584def generate(prompt):585 return pipe(prompt).images586```587 588**Expected Performance Gains:**589- Basic AoT: 1.3x-1.8x speedup590- + FP8 Quantization: Additional 1.2x speedup 591- + FlashAttention-3: Additional attention speedup592- Total potential: 2x-3x faster inference593 594**Hardware Requirements:**595- FP8 quantization requires CUDA compute capability โฅ 9.0 (H200 โ
)596- FlashAttention-3 works on H200 hardware via kernels library597- Dynamic shapes add flexibility for variable input sizes598 599## Complete Gradio API Reference600 601This reference is automatically synced from https://www.gradio.app/llms.txt to ensure accuracy.602 603"""604 605 # Search-enabled prompt606 search_prompt = """You are an expert Gradio developer with access to real-time web search. Create a complete, working Gradio application based on the user's request. When needed, use web search to find current best practices or verify latest Gradio features. Generate all necessary code to make the application functional and runnable.607 608๐จ IMPORTANT: If the user is asking to use external APIs (like OpenRouter, OpenAI API, Hugging Face Inference API, etc.), DO NOT use @spaces.GPU decorators or any ZeroGPU features. External APIs handle the model inference remotely, so GPU allocation on the Spaces instance is not needed.609 610๐จ CRITICAL REQUIREMENT: If the user provides ANY diffusion model code (FLUX, Stable Diffusion, etc.) that runs locally (not via API), you MUST implement ZeroGPU ahead-of-time (AoT) compilation. This is mandatory and provides 1.3x-1.8x performance improvements. Do not create basic Gradio apps without AoT optimization for diffusion models.611 612## ZeroGPU Integration (MANDATORY)613 614ALWAYS use ZeroGPU for GPU-dependent functions in Gradio apps:615 6161. Import the spaces module: `import spaces`6172. Decorate GPU-dependent functions with `@spaces.GPU`6183. Specify appropriate duration based on expected runtime:619 - Quick inference (< 30s): `@spaces.GPU(duration=30)`620 - Standard generation (30-60s): `@spaces.GPU` (default 60s)621 - Complex generation (60-120s): `@spaces.GPU(duration=120)`622 - Heavy processing (120-180s): `@spaces.GPU(duration=180)`623 624Example usage:625```python626import spaces627from diffusers import DiffusionPipeline628 629pipe = DiffusionPipeline.from_pretrained(...)630pipe.to('cuda')631 632@spaces.GPU(duration=120)633def generate(prompt):634 return pipe(prompt).images635 636gr.Interface(637 fn=generate,638 inputs=gr.Text(),639 outputs=gr.Gallery(),640).launch()641```642 643Duration Guidelines:644- Shorter durations improve queue priority for users645- Text-to-image: typically 30-60 seconds646- Image-to-image: typically 20-40 seconds 647- Video generation: typically 60-180 seconds648- Audio/music generation: typically 30-90 seconds649- Model loading + inference: add 10-30s buffer650- AoT compilation during startup: use @spaces.GPU(duration=1500) for maximum allowed duration651 652Functions that typically need @spaces.GPU:653- Image generation (text-to-image, image-to-image)654- Video generation655- Audio/music generation656- Model inference with transformers, diffusers657- Any function using .to('cuda') or GPU operations658 659## CRITICAL: Use ZeroGPU AoT Compilation for ALL Diffusion Models660 661FOR ANY DIFFUSION MODEL (FLUX, Stable Diffusion, etc.), YOU MUST IMPLEMENT AHEAD-OF-TIME COMPILATION.662This is NOT optional - it provides 1.3x-1.8x speedup and is essential for production ZeroGPU Spaces.663 664ALWAYS implement this pattern for diffusion models:665 666### MANDATORY: Basic AoT Compilation Pattern667YOU MUST USE THIS EXACT PATTERN for any diffusion model (FLUX, Stable Diffusion, etc.):668 6691. ALWAYS add AoT compilation function with @spaces.GPU(duration=1500)6702. ALWAYS use spaces.aoti_capture to capture inputs6713. ALWAYS use torch.export.export to export the transformer6724. ALWAYS use spaces.aoti_compile to compile6735. ALWAYS use spaces.aoti_apply to apply to pipeline674 675### Required AoT Implementation676 677For production Spaces with heavy models, use ahead-of-time (AoT) compilation for 1.3x-1.8x speedups:678 679### Basic AoT Compilation680```python681import spaces682import torch683from diffusers import DiffusionPipeline684 685MODEL_ID = 'black-forest-labs/FLUX.1-dev'686pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)687pipe.to('cuda')688 689@spaces.GPU(duration=1500) # Maximum duration allowed during startup690def compile_transformer():691 # 1. Capture example inputs692 with spaces.aoti_capture(pipe.transformer) as call:693 pipe("arbitrary example prompt")694 695 # 2. Export the model696 exported = torch.export.export(697 pipe.transformer,698 args=call.args,699 kwargs=call.kwargs,700 )701 702 # 3. Compile the exported model703 return spaces.aoti_compile(exported)704 705# 4. Apply compiled model to pipeline706compiled_transformer = compile_transformer()707spaces.aoti_apply(compiled_transformer, pipe.transformer)708 709@spaces.GPU710def generate(prompt):711 return pipe(prompt).images712```713 714### Advanced Optimizations715 716#### FP8 Quantization (Additional 1.2x speedup on H200)717```python718from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig719 720@spaces.GPU(duration=1500)721def compile_transformer_with_quantization():722 # Quantize before export for FP8 speedup723 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())724 725 with spaces.aoti_capture(pipe.transformer) as call:726 pipe("arbitrary example prompt")727 728 exported = torch.export.export(729 pipe.transformer,730 args=call.args,731 kwargs=call.kwargs,732 )733 return spaces.aoti_compile(exported)734```735 736#### Dynamic Shapes (Variable input sizes)737```python738from torch.utils._pytree import tree_map739 740@spaces.GPU(duration=1500)741def compile_transformer_dynamic():742 with spaces.aoti_capture(pipe.transformer) as call:743 pipe("arbitrary example prompt")744 745 # Define dynamic dimension ranges (model-dependent)746 transformer_hidden_dim = torch.export.Dim('hidden', min=4096, max=8212)747 748 # Map argument names to dynamic dimensions749 transformer_dynamic_shapes = {750 "hidden_states": {1: transformer_hidden_dim}, 751 "img_ids": {0: transformer_hidden_dim},752 }753 754 # Create dynamic shapes structure755 dynamic_shapes = tree_map(lambda v: None, call.kwargs)756 dynamic_shapes.update(transformer_dynamic_shapes)757 758 exported = torch.export.export(759 pipe.transformer,760 args=call.args,761 kwargs=call.kwargs,762 dynamic_shapes=dynamic_shapes,763 )764 return spaces.aoti_compile(exported)765```766 767#### Multi-Compile for Different Resolutions768```python769@spaces.GPU(duration=1500)770def compile_multiple_resolutions():771 compiled_models = {}772 resolutions = [(512, 512), (768, 768), (1024, 1024)]773 774 for width, height in resolutions:775 # Capture inputs for specific resolution776 with spaces.aoti_capture(pipe.transformer) as call:777 pipe(f"test prompt {width}x{height}", width=width, height=height)778 779 exported = torch.export.export(780 pipe.transformer,781 args=call.args,782 kwargs=call.kwargs,783 )784 compiled_models[f"{width}x{height}"] = spaces.aoti_compile(exported)785 786 return compiled_models787 788# Usage with resolution dispatch789compiled_models = compile_multiple_resolutions()790 791@spaces.GPU792def generate_with_resolution(prompt, width=1024, height=1024):793 resolution_key = f"{width}x{height}"794 if resolution_key in compiled_models:795 # Temporarily apply the right compiled model796 spaces.aoti_apply(compiled_models[resolution_key], pipe.transformer)797 return pipe(prompt, width=width, height=height).images798```799 800#### FlashAttention-3 Integration801```python802from kernels import get_kernel803 804# Load pre-built FA3 kernel compatible with H200805try:806 vllm_flash_attn3 = get_kernel("kernels-community/vllm-flash-attn3")807 print("โ
FlashAttention-3 kernel loaded successfully")808except Exception as e:809 print(f"โ ๏ธ FlashAttention-3 not available: {e}")810 811# Custom attention processor example812class FlashAttention3Processor:813 def __call__(self, attn, hidden_states, encoder_hidden_states=None, attention_mask=None):814 # Use FA3 kernel for attention computation815 return vllm_flash_attn3(hidden_states, encoder_hidden_states, attention_mask)816 817# Apply FA3 processor to model818if 'vllm_flash_attn3' in locals():819 for name, module in pipe.transformer.named_modules():820 if hasattr(module, 'processor'):821 module.processor = FlashAttention3Processor()822```823 824### Complete Optimized Example825```python826import spaces827import torch828from diffusers import DiffusionPipeline829from torchao.quantization import quantize_, Float8DynamicActivationFloat8WeightConfig830 831MODEL_ID = 'black-forest-labs/FLUX.1-dev'832pipe = DiffusionPipeline.from_pretrained(MODEL_ID, torch_dtype=torch.bfloat16)833pipe.to('cuda')834 835@spaces.GPU(duration=1500)836def compile_optimized_transformer():837 # Apply FP8 quantization838 quantize_(pipe.transformer, Float8DynamicActivationFloat8WeightConfig())839 840 # Capture inputs841 with spaces.aoti_capture(pipe.transformer) as call:842 pipe("optimization test prompt")843 844 # Export and compile845 exported = torch.export.export(846 pipe.transformer,847 args=call.args,848 kwargs=call.kwargs,849 )850 return spaces.aoti_compile(exported)851 852# Compile during startup853compiled_transformer = compile_optimized_transformer()854spaces.aoti_apply(compiled_transformer, pipe.transformer)855 856@spaces.GPU857def generate(prompt):858 return pipe(prompt).images859```860 861**Expected Performance Gains:**862- Basic AoT: 1.3x-1.8x speedup863- + FP8 Quantization: Additional 1.2x speedup 864- + FlashAttention-3: Additional attention speedup865- Total potential: 2x-3x faster inference866 867**Hardware Requirements:**868- FP8 quantization requires CUDA compute capability โฅ 9.0 (H200 โ
)869- FlashAttention-3 works on H200 hardware via kernels library870- Dynamic shapes add flexibility for variable input sizes871 872## Complete Gradio API Reference873 874This reference is automatically synced from https://www.gradio.app/llms.txt to ensure accuracy.875 876"""877 878 # Update the prompts879 GRADIO_SYSTEM_PROMPT = base_prompt + docs_content + "\n\nAlways use the exact function signatures from this API reference and follow modern Gradio patterns.\n\nIMPORTANT: Always include \"Built with anycoder\" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder"880 GRADIO_SYSTEM_PROMPT_WITH_SEARCH = search_prompt + docs_content + "\n\nAlways use the exact function signatures from this API reference and follow modern Gradio patterns.\n\nIMPORTANT: Always include \"Built with anycoder\" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder"881 882def update_json_system_prompts():883 """Update the global JSON system prompts with latest ComfyUI documentation"""884 global JSON_SYSTEM_PROMPT, JSON_SYSTEM_PROMPT_WITH_SEARCH885 886 docs_content = get_comfyui_docs_content()887 888 # Base system prompt889 base_prompt = """You are an expert JSON developer. Generate clean, valid JSON data based on the user's request. Follow JSON syntax rules strictly:890- Use double quotes for strings891- No trailing commas892- Proper nesting and structure893- Valid data types (string, number, boolean, null, object, array)894 895Generate ONLY the JSON data requested - no HTML, no applications, no explanations outside the JSON. The output should be pure, valid JSON that can be parsed directly.896 897"""898 899 # Search-enabled system prompt900 search_prompt = """You are an expert JSON developer. You have access to real-time web search. When needed, use web search to find the latest information or data structures for your JSON generation.901 902Generate clean, valid JSON data based on the user's request. Follow JSON syntax rules strictly:903- Use double quotes for strings904- No trailing commas905- Proper nesting and structure906- Valid data types (string, number, boolean, null, object, array)907 908Generate ONLY the JSON data requested - no HTML, no applications, no explanations outside the JSON. The output should be pure, valid JSON that can be parsed directly.909 910"""911 912 # Add ComfyUI documentation if available913 if docs_content.strip():914 comfyui_section = f"""915## ComfyUI Reference Documentation916 917When generating JSON data related to ComfyUI workflows, nodes, or configurations, use this reference:918 919{docs_content}920 921This reference is automatically synced from https://docs.comfy.org/llms.txt to ensure accuracy.922 923"""924 base_prompt += comfyui_section925 search_prompt += comfyui_section926 927 # Update the prompts928 JSON_SYSTEM_PROMPT = base_prompt929 JSON_SYSTEM_PROMPT_WITH_SEARCH = search_prompt930 931# Initialize Gradio documentation on startup932def initialize_gradio_docs():933 """Initialize Gradio documentation on application startup"""934 try:935 update_gradio_system_prompts()936 if should_update_gradio_docs():937 print("๐ Gradio documentation system initialized (fetched fresh content)")938 else:939 print("๐ Gradio documentation system initialized (using cached content)")940 except Exception as e:941 print(f"Warning: Failed to initialize Gradio documentation: {e}")942 943# Initialize ComfyUI documentation on startup944def initialize_comfyui_docs():945 """Initialize ComfyUI documentation on application startup"""946 try:947 update_json_system_prompts()948 if should_update_comfyui_docs():949 print("๐ ComfyUI documentation system initialized (fetched fresh content)")950 else:951 print("๐ ComfyUI documentation system initialized (using cached content)")952 except Exception as e:953 print(f"Warning: Failed to initialize ComfyUI documentation: {e}")954 955# Configuration956HTML_SYSTEM_PROMPT = """ONLY USE HTML, CSS AND JAVASCRIPT. If you want to use ICON make sure to import the library first. Try to create the best UI possible by using only HTML, CSS and JAVASCRIPT. MAKE IT RESPONSIVE USING MODERN CSS. Use as much as you can modern CSS for the styling, if you can't do something with modern CSS, then use custom CSS. Also, try to elaborate as much as you can, to create something unique. ALWAYS GIVE THE RESPONSE INTO A SINGLE HTML FILE957 958For website redesign tasks:959- Use the provided original HTML code as the starting point for redesign960- Preserve all original content, structure, and functionality961- Keep the same semantic HTML structure but enhance the styling962- Reuse all original images and their URLs from the HTML code963- Create a modern, responsive design with improved typography and spacing964- Use modern CSS frameworks and design patterns965- Ensure accessibility and mobile responsiveness966- Maintain the same navigation and user flow967- Enhance the visual design while keeping the original layout structure968 969If an image is provided, analyze it and use the visual information to better understand the user's requirements.970 971Always respond with code that can be executed or rendered directly.972 973Generate complete, working HTML code that can be run immediately.974 975IMPORTANT: Always include "Built with anycoder" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder"""976 977def validate_video_html(video_html: str) -> bool:978 """Validate that the video HTML is well-formed and safe to insert."""979 try:980 # Basic checks for video HTML structure981 if not video_html or not video_html.strip():982 return False983 984 # Check for required video elements985 if '<video' not in video_html or '</video>' not in video_html:986 return False987 988 # Check for proper source tag989 if '<source' not in video_html:990 return False991 992 # Check for valid video source (data URI, HF URL, or file URL)993 has_data_uri = 'data:video/mp4;base64,' in video_html994 has_hf_url = 'https://huggingface.co/datasets/' in video_html and '/resolve/main/' in video_html995 has_file_url = 'file://' in video_html996 if not (has_data_uri or has_hf_url or has_file_url):997 return False998 999 # Basic HTML structure validation1000 video_start = video_html.find('<video')1001 video_end = video_html.find('</video>') + 81002 if video_start == -1 or video_end == 7: # 7 means </video> not found1003 return False1004 1005 return True1006 except Exception:1007 return False1008 1009def llm_place_media(html_content: str, media_html_tag: str, media_kind: str = "image") -> str:1010 """Ask a lightweight model to produce search/replace blocks that insert media_html_tag in the best spot.1011 1012 The model must return ONLY our block format using SEARCH_START/DIVIDER/REPLACE_END.1013 """1014 try:1015 client = get_inference_client("Qwen/Qwen3-Coder-480B-A35B-Instruct", "auto")1016 system_prompt = (1017 "You are a code editor. Insert the provided media tag into the given HTML in the most semantically appropriate place.\n"1018 "For video elements: prefer replacing placeholder images or inserting in hero sections with proper container divs.\n"1019 "For image elements: prefer replacing placeholder images or inserting near related content.\n"1020 "CRITICAL: Ensure proper HTML structure - videos should be wrapped in appropriate containers.\n"1021 "Return ONLY search/replace blocks using the exact markers: <<<<<<< SEARCH, =======, >>>>>>> REPLACE.\n"1022 "Do NOT include any commentary. Ensure the SEARCH block matches exact lines from the input.\n"1023 "When inserting videos, ensure they are properly contained within semantic HTML elements.\n"1024 )1025 # Truncate very long media tags for LLM prompt only to prevent token limits1026 truncated_media_tag_for_prompt = media_html_tag1027 if len(media_html_tag) > 2000:1028 # For very long data URIs, show structure but truncate the data for LLM prompt1029 if 'data:video/mp4;base64,' in media_html_tag:1030 start_idx = media_html_tag.find('data:video/mp4;base64,')1031 end_idx = media_html_tag.find('"', start_idx)1032 if start_idx != -1 and end_idx != -1:1033 truncated_media_tag_for_prompt = (1034 media_html_tag[:start_idx] + 1035 'data:video/mp4;base64,[TRUNCATED_BASE64_DATA]' + 1036 media_html_tag[end_idx:]1037 )1038 1039 user_payload = (1040 "HTML Document:\n" + html_content + "\n\n" +1041 f"Media ({media_kind}):\n" + truncated_media_tag_for_prompt + "\n\n" +1042 "Produce search/replace blocks now."1043 )1044 messages = [1045 {"role": "system", "content": system_prompt},1046 {"role": "user", "content": user_payload},1047 ]1048 completion = client.chat.completions.create(1049 model="Qwen/Qwen3-Coder-480B-A35B-Instruct",1050 messages=messages,1051 max_tokens=2000,1052 temperature=0.2,1053 )1054 text = (completion.choices[0].message.content or "") if completion and completion.choices else ""1055 1056 # Replace any truncated placeholders with the original full media HTML1057 if '[TRUNCATED_BASE64_DATA]' in text and 'data:video/mp4;base64,[TRUNCATED_BASE64_DATA]' in truncated_media_tag_for_prompt:1058 # Extract the original base64 data from the full media tag1059 original_start = media_html_tag.find('data:video/mp4;base64,')1060 original_end = media_html_tag.find('"', original_start)1061 if original_start != -1 and original_end != -1:1062 original_data_uri = media_html_tag[original_start:original_end]1063 text = text.replace('data:video/mp4;base64,[TRUNCATED_BASE64_DATA]', original_data_uri)1064 1065 return text.strip()1066 except Exception as e:1067 print(f"[LLMPlaceMedia] Fallback due to error: {e}")1068 return ""1069 1070# Stricter prompt for GLM-4.5V to ensure a complete, runnable HTML document with no escaped characters1071GLM45V_HTML_SYSTEM_PROMPT = """You are an expert front-end developer.1072 1073Output a COMPLETE, STANDALONE HTML document that renders directly in a browser.1074 1075Hard constraints:1076- DO NOT use React, ReactDOM, JSX, Babel, Vue, Angular, Svelte, or any SPA framework.1077- Use ONLY plain HTML, CSS, and vanilla JavaScript.1078- Allowed external resources: Tailwind CSS CDN, Font Awesome CDN, Google Fonts.1079- Do NOT escape characters (no \\n, \\t, or escaped quotes). Output raw HTML/JS/CSS.1080 1081Structural requirements:1082- Include <!DOCTYPE html>, <html>, <head>, and <body> with proper nesting1083- Include required <link> tags for any CSS you reference (e.g., Tailwind, Font Awesome, Google Fonts)1084- Keep everything in ONE file; inline CSS/JS as needed1085 1086Generate complete, working HTML code that can be run immediately.1087 1088IMPORTANT: Always include "Built with anycoder" as clickable text in the header/top section of your application that links to https://huggingface.co/spaces/akhaliq/anycoder1089"""1090 1091# ---------------------------------------------------------------------------1092# Video temp-file management (per-session tracking and cleanup)1093# ---------------------------------------------------------------------------1094VIDEO_TEMP_DIR = os.path.join(tempfile.gettempdir(), "anycoder_videos")1095VIDEO_FILE_TTL_SECONDS = 6 * 60 * 60 # 6 hours1096_SESSION_VIDEO_FILES: Dict[str, List[str]] = {}1097_VIDEO_FILES_LOCK = threading.Lock()1098 1099 1100def _ensure_video_dir_exists() -> None:1101 try:1102 os.makedirs(VIDEO_TEMP_DIR, exist_ok=True)1103 except Exception:1104 pass1105 1106 1107def _register_video_for_session(session_id: str | None, file_path: str) -> None:1108 if not session_id or not file_path:1109 return1110 with _VIDEO_FILES_LOCK:1111 if session_id not in _SESSION_VIDEO_FILES:1112 _SESSION_VIDEO_FILES[session_id] = []1113 _SESSION_VIDEO_FILES[session_id].append(file_path)1114 1115 1116def cleanup_session_videos(session_id: str | None) -> None:1117 if not session_id:1118 return1119 with _VIDEO_FILES_LOCK:1120 file_list = _SESSION_VIDEO_FILES.pop(session_id, [])1121 for path in file_list:1122 try:1123 if path and os.path.exists(path):1124 os.unlink(path)1125 except Exception:1126 # Best-effort cleanup1127 pass1128 1129 1130def reap_old_videos(ttl_seconds: int = VIDEO_FILE_TTL_SECONDS) -> None:1131 """Delete old video files in the temp directory based on modification time."""1132 try:1133 _ensure_video_dir_exists()1134 now_ts = time.time()1135 for name in os.listdir(VIDEO_TEMP_DIR):1136 path = os.path.join(VIDEO_TEMP_DIR, name)1137 try:1138 if not os.path.isfile(path):1139 continue1140 mtime = os.path.getmtime(path)1141 if now_ts - mtime > ttl_seconds:1142 os.unlink(path)1143 except Exception:1144 pass1145 except Exception:1146 # Temp dir might not exist or be accessible; ignore1147 pass1148 1149# ---------------------------------------------------------------------------1150# Audio temp-file management (per-session tracking and cleanup)1151# ---------------------------------------------------------------------------1152AUDIO_TEMP_DIR = os.path.join(tempfile.gettempdir(), "anycoder_audio")1153AUDIO_FILE_TTL_SECONDS = 6 * 60 * 60 # 6 hours1154_SESSION_AUDIO_FILES: Dict[str, List[str]] = {}1155_AUDIO_FILES_LOCK = threading.Lock()1156 1157 1158def _ensure_audio_dir_exists() -> None:1159 try:1160 os.makedirs(AUDIO_TEMP_DIR, exist_ok=True)1161 except Exception:1162 pass1163 1164 1165def _register_audio_for_session(session_id: str | None, file_path: str) -> None:1166 if not session_id or not file_path:1167 return1168 with _AUDIO_FILES_LOCK:1169 if session_id not in _SESSION_AUDIO_FILES:1170 _SESSION_AUDIO_FILES[session_id] = []1171 _SESSION_AUDIO_FILES[session_id].append(file_path)1172 1173 1174def cleanup_session_audio(session_id: str | None) -> None:1175 if not session_id:1176 return1177 with _AUDIO_FILES_LOCK:1178 file_list = _SESSION_AUDIO_FILES.pop(session_id, [])1179 for path in file_list:1180 try:1181 if path and os.path.exists(path):1182 os.unlink(path)1183 except Exception:1184 pass1185 1186 1187def reap_old_audio(ttl_seconds: int = AUDIO_FILE_TTL_SECONDS) -> None:1188 try:1189 _ensure_audio_dir_exists()1190 now_ts = time.time()1191 for name in os.listdir(AUDIO_TEMP_DIR):1192 path = os.path.join(AUDIO_TEMP_DIR, name)1193 try:1194 if not os.path.isfile(path):1195 continue1196 mtime = os.path.getmtime(path)1197 if now_ts - mtime > ttl_seconds:1198 os.unlink(path)1199 except Exception:1200 pass