Team Ai
Apppublic

Csplk/moondream2-batch-processing

sourceHugging Faceapache-2.0updated 7mo agoView on Hugging Face
8likes
designbatch.py247 linesDownload Raw Back to root
1import os2import torch3import gradio as gr4from transformers import AutoTokenizer, AutoModelForCausalLM, pipeline5from PIL import Image6import requests7import json8import base649from io import BytesIO10 11# Check for CUDA availability for PyTorch12if torch.cuda.is_available():13    device, dtype = "cuda", torch.bfloat1614else:15    device, dtype = "cpu", torch.float3216 17# Load Moondream3 Preview for image analysis18moondream3_model_id = "moondream/moondream3-preview"19tokenizer_moondream3 = AutoTokenizer.from_pretrained(moondream3_model_id)20moondream3 = AutoModelForCausalLM.from_pretrained(21    moondream3_model_id,22    trust_remote_code=True,23    torch_dtype=dtype,24    device_map={"": device}25).eval()26moondream3.compile()  # Optional: speeds up inference27 28# Initialize DeepSeek-V2 for chat completion29deepseek_model_name = "deepseek-ai/DeepSeek-V2"30tokenizer_deepseek = AutoTokenizer.from_pretrained(deepseek_model_name)31deepseek_model = AutoModelForCausalLM.from_pretrained(32    deepseek_model_name,33    torch_dtype=torch.bfloat16,34    device_map="auto"35)36chat_pipe = pipeline(37    "text-generation",38    model=deepseek_model,39    tokenizer=tokenizer_deepseek,40    max_new_tokens=512,41    temperature=0.7,42    top_p=0.9,43    repetition_penalty=1.1,44    do_sample=True,45)46 47def deepseek_chat(user_message, is_json=False):48    """Chat completion using DeepSeek-V2."""49    prompt = f"<|BeginOfUtterance|>User: {user_message}<|EndOfUtterance|><|BeginOfUtterance|>Assistant:"50    response = chat_pipe(prompt, return_full_text=False)[0]["generated_text"]51    assistant_response = response.split("<|BeginOfUtterance|>Assistant:")[-1].strip()52    return assistant_response53 54# Extract features from images using Moondream355def extract_features(image_tuples):56    headers = ["Image", "Layout", "Decor", "Atmosphere", "Lighting", "Color scheme", "Furniture style"]57    data = []58 59    image_embeds = [img[0] for img in image_tuples if img[0] is not None]60    prompts = [61        "Describe the spatial arrangement of furniture, walls, and other elements in this image.",62        "What type, style, and arrangement of decorative elements are present in this image?",63        "What mood, ambiance, and overall feeling does this image evoke?",64        "What type, intensity, placement, and direction of lighting is present in this image?",65        "What are the dominant colors, color palette, and color harmony in this image?",66        "What type, shape, material, and arrangement of furniture is present in this image?"67    ]68 69    answers = []70    for prompt in prompts:71        image_answers = moondream3.batch_answer(72            images=[img.convert("RGB") for img in image_embeds],73            prompts=[prompt] * len(image_embeds),74            tokenizer=tokenizer_moondream3,75        )76        answers.append(image_answers)77 78    for i in range(len(image_tuples)):79        image_name = f"image{i+1}"80        image_answers = [answer[i] for answer in answers]81        print(f"image{i+1}_answers \n {image_answers} \n")82        data.append([image_name] + image_answers)83 84    result = {'headers': headers, 'data': data}85    return result86 87# Describe room from image using Moondream388def describe_room(image):89    headers = ["Image", "Layout", "Decor", "Atmosphere", "Lighting", "Color scheme", "Furniture style"]90    data = []91 92    image_embeds = [image.convert("RGB")] * 693    prompts = [94        "Describe the spatial arrangement of furniture, walls, and other elements in this image.",95        "What type, style, and arrangement of decorative elements are present in this image?",96        "What mood, ambiance, and overall feeling does this image evoke?",97        "What type, intensity, placement, and direction of lighting is present in this image?",98        "What are the dominant colors, color palette, and color harmony in this image?",99        "What type, shape, material, and arrangement of furniture is present in this image?"100    ]101 102    answers = moondream3.batch_answer(103        images=image_embeds,104        prompts=prompts,105        tokenizer=tokenizer_moondream3,106    )107 108    image_name = "ClientRoom"109    print(f"ClientRoom_answers \n {answers} \n")110    data.append([image_name] + answers)111 112    result = {'headers': headers, 'data': data}113    return result114 115def merge_features(inspiration_features):116    preferenec_map_extraction = f"""117You are one of the worlds most knowledgeable minds in the field of both theoretical and applied interior design.118- You are detailed119- You are meticulous120- You can distil a large potentially unstructured potentially multimodal range of input data sources into a highly accurate all encompassing representation of the interior design concept preferences of the input source by mapping input data using a model of fundamental interior design component definitions121- You can come up with professionally structured, fully detailed, well thought out and all encompassing applied interior design proposals from initial conceptualization and planning to a complete and finished interior design of real world space122- Generally you can help answer any question or assist in any task asked of you relating to anything in the realm of applied and theoretical design and interior design123 124Your task is to analyze the interior design style given information after <<<>>> and merge the analysis results together to generate a comprehensive design style preference map representation for the user who uploaded some images. Return as JSON125 126<<<127{inspiration_features}128>>>129"""130    print(f"\npreference_map_extraction prompt\n{preferenec_map_extraction}\n")131    prefmap = deepseek_chat(preferenec_map_extraction, is_json=True)132    print(f"\n merge_features chat_response\n{prefmap}\n")133    return prefmap134 135def create_design_concept_report(room_description, inspiration_features):136    design_report_prompt = f"""137    Generate a detailed interior design plan proposal report structured as markdown138    - The report should include three design plan concepts for the clients space based on the clients interior design component preference representation generated from the inspirational images they uploaded clients room that is the target of the project and the design preference map generated from the inspirational design images they uploaded139    - The report should have an introduction, sections on Style Preference, Color Scheme, Furniture Style, Lighting, Atmosphere, Decor, and Layout for each concept, as well as a placeholder for a mood board image starting each concept section.140    - Finally, the report should have a summary to conclude the design plan.141 142    Very detailed information about the clients room based on the photo they uploaded:143    {room_description}144 145    Design preference map generated from the inspirational design images they uploaded:146    {inspiration_features}147"""148    print(f"\ndesign_report_prompt\n{design_report_prompt}\n")149    designreport = deepseek_chat(design_report_prompt)150    print(f"\ndesign concept chat_response\n{designreport}\n")151    return designreport152 153def queryllm(payload):154    response = requests.post(textgen_API_URL, headers=headers, json=payload)155    print(response)156    return response.json()157 158def generate_mood_board_image(prompt):159    payload = {"inputs": prompt}160    response = requests.post(texttoimage_API_URL, headers=headers, json=payload)161    return response.content162 163def getmoodboardprompts(designreport):164    mood_board_descriptions_prompt = f"""165    ### interior design report plan166    {designreport}167    ###168 169    Generate a text prompt for each of the interior design concepts described in the interior design report plan that can be sent to a text-to-image model and receive a design project mood board.170    The prompt should clearly describe what should go onto the moodboard for each design concept and be structured JSON. For example:171    {{172        "Concept1": "Create a mood board for a modern cozy retreat bedroom with a warm and inviting atmosphere. Include a white and brown color palette, modern and contemporary furniture with clean lines, a cozy and functional bed, nightstands with elegant designs, a bench at the foot of the bed with storage, sheer curtains on the window, floor lamps and table lamps with layered lighting effects, potted plants, a vase with branches and twigs, a bowl, a clock, and books on the nightstands.",173        "Concept2": "Create a mood board for another concept..."174    }}175    Only output the JSON, nothing else, no explanations or commentary.176    """177    print(f"\nmood_board_descriptions_prompt:\n{mood_board_descriptions_prompt}\n")178    mood_board_descriptions = deepseek_chat(mood_board_descriptions_prompt)179    print(f"\nmood_board_descriptions_prompt chat_response\n{mood_board_descriptions}\n")180    return json.loads(mood_board_descriptions)181 182def generate_moodboards(mb_prompts):183    moodboard_images = {}184    for concept, prompt in mb_prompts.items():185        image_data = generate_mood_board_image(prompt)186        file_path = f"moodboard_{concept}.jpg"187        with open(file_path, "wb") as f:188            f.write(image_data)189        moodboard_images[concept] = file_path190    return moodboard_images191 192def add_moodboards_to_report(moodboard_images, report):193    add_moodboards_prompt = f"""194    mood board images195    <<<196    {moodboard_images}197    >>>198 199    report200    <<<201    {report}202    >>>203 204    Insert paths for each mood board image into the respective placeholder for each in the report and respond with the revised report with moodboard images inserted only, no explanations or commentary205    """206    print(f"\nadd_moodboards_prompt\n{add_moodboards_prompt}\n")207    revised_report = deepseek_chat(add_moodboards_prompt)208    print(f"\nrevised_report\n{revised_report}\n")209    return revised_report210 211# Gradio Interface212def process_images(design_images, room_image):213    design_descriptions = extract_features(design_images)214    room_description = describe_room(room_image)215 216    preference_map = merge_features(design_descriptions)217    print(f"\npreference_map\n{preference_map}\n")218 219    design_report = create_design_concept_report(room_description, preference_map)220    print(f"\ndesign_report\n{design_report}\n")221 222    mb_prompts = getmoodboardprompts(design_report)223    print(f"\nmb_prompts\n{mb_prompts}\n")224 225    moodboard_images = generate_moodboards(mb_prompts)226    print(f"\nmoodboard_images\n{moodboard_images}\n")227 228    revised_report = add_moodboards_to_report(moodboard_images, design_report)229    print("revised_report")230    print(revised_report)231    print("preference map")232    print(preference_map)233    return revised_report, preference_map234 235gallery = gr.components.Gallery(label="Upload Images of Preferred Design Styles", type="pil")236image_input = gr.components.Image(label="Upload Image of Your Room", type="pil")237report_output = gr.components.Markdown(label="Design Concept Report with Mood Boards")238json_output = gr.components.JSON(label="Design Preference Map")239 240interface = gr.Interface(241    fn=process_images,242    inputs=[gallery, image_input],243    outputs=[report_output, json_output],244    title="Interior Design Assistant",245    description="Upload images of your preferred interior design styles and a photo of your room to receive a custom design concept report and preference map."246)247