Team Ai
Apppublic

Hasani/Binary-Image-Classification-In-The-Wild

sourceHugging Faceopenrailupdated 3y agoView on Hugging Face
1likes
app.py59 linesDownload Raw Back to root
1from PIL import Image2from transformers import CLIPProcessor, CLIPModel3import gradio as gr4import torchvision.transforms as transforms5 6# Initialize CLIP model and processor7processor = CLIPProcessor.from_pretrained("openai/clip-vit-base-patch32")8model = CLIPModel.from_pretrained("openai/clip-vit-base-patch32")9 10def image_similarity(image: Image.Image, positive_prompt: str, negative_prompts: str):11    # Convert the PIL Image to a tensor and preprocess12    transform = transforms.Compose([13        transforms.Resize((224, 224)),14        transforms.ToTensor(),15        transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5)),16    ])17    image_tensor = transform(image).unsqueeze(0)  # Add batch dimension18 19    # Split the negative prompts string into a list of prompts20    negative_prompts_list = negative_prompts.split(";")21    # Combine positive and negative prompts into one list22    prompts = [positive_prompt.strip()] + [np.strip() for np in negative_prompts_list]23 24    # Process prompts and image tensor25    inputs = processor(26        text=prompts,27        images=image_tensor,28        return_tensors="pt",29        padding=True30    )31 32    outputs = model(**inputs)33    logits_per_image = outputs.logits_per_image34    probs = logits_per_image.softmax(dim=1)35 36    # Determine if positive prompt has a higher probability than any of the negative prompts37    is_positive_highest = probs[0][0] > max(probs[0][1:])38 39    return bool(is_positive_highest), f"Probability for Positive Prompt: {probs[0][0]:.4f}"40 41interface = gr.Interface(42    fn=image_similarity, 43    inputs=[44        gr.components.Image(type="pil"), 45        gr.components.Text(label="Enter positive prompt e.g. 'a person drinking a beverage'"),46        gr.components.Textbox(label="Enter negative prompts, separated by semicolon e.g. 'an empty scene; person without beverage'", placeholder="negative prompt 1; negative prompt 2; ..."),47    ], 48    outputs=[49        gr.components.Textbox(label="Result"),50        gr.components.Textbox(label="Probability for Positive Prompt")51    ],52    title="Engagify's Image Action Detection",53    description="[Author: Ibrahim Hasani] This Method uses CLIP-VIT [Version: BASE-PATCH-16] to determine if an action is being performed in an image or not. (Binary Classifier). It contrasts an Action against multiple negative labels. Ensure the prompts accurately describe the desired detection.",54    live=False,55    theme=gr.themes.Monochrome(),56 57)58 59interface.launch()