ImageProcessing/backend
0
1from PIL import Image
2import io
3from transformers import AutoTokenizer, CLIPProcessor, CLIPModel
4import torch
5
6# Load CLIP model and processor
7model_name = "openai/clip-vit-base-patch32"
8loaded_model = CLIPModel.from_pretrained(model_name)
9loaded_processor = CLIPProcessor.from_pretrained(model_name)
10
11def getTextEmbedding(text):
12 # Preprocess the text
13 print("tear")
14 inputs_text = loaded_processor(text=[text], return_tensors="pt", padding=True)
15 print("here")
16 # Forward pass through the model
17 with torch.no_grad():
18 # Get the text features
19 text_features = loaded_model.get_text_features(input_ids=inputs_text.input_ids, attention_mask=inputs_text.attention_mask)
20 print("bear")
21 # Convert tensor to numpy array for better readability
22 text_embedding = text_features.squeeze().numpy()
23 print("done")
24 return text_embedding
25
26def getImageEmbedding(binary_image_data):
27 # Load and preprocess the image
28 image = Image.open(io.BytesIO(binary_image_data))
29 inputs = loaded_processor(images=image, return_tensors="pt", padding=True)
30
31 # Forward pass through the model
32 with torch.no_grad():
33 # Get the image features
34 image_features = loaded_model.get_image_features(pixel_values=inputs.pixel_values)
35
36 # Convert tensor to numpy array for better readability
37 image_embedding = image_features.squeeze().numpy()
38
39 return image_embedding
40
41 