Team Ai
Apppublic

Gosula/Stable_diffusion_model

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
device.py44 linesDownload Raw Back to root
1from base64 import b64encode2import numpy3import torch4from diffusers import AutoencoderKL, LMSDiscreteScheduler, UNet2DConditionModel5from huggingface_hub import notebook_login6 7# For video display:8from IPython.display import HTML9from matplotlib import pyplot as plt10from pathlib import Path11from PIL import Image12from torch import autocast13from torchvision import transforms as tfms14from tqdm.auto import tqdm15from transformers import CLIPTextModel, CLIPTokenizer, logging16import os17torch_device = "cuda" if torch.cuda.is_available() else "mps" if torch.backends.mps.is_available() else "cpu"18#torch_device = "cpu" 19 20 21# Load the autoencoder model which will be used to decode the latents into image space.22vae = AutoencoderKL.from_pretrained("CompVis/stable-diffusion-v1-4", subfolder="vae")23 24# Load the tokenizer and text encoder to tokenize and encode the text.25tokenizer = CLIPTokenizer.from_pretrained("openai/clip-vit-large-patch14")26text_encoder = CLIPTextModel.from_pretrained("openai/clip-vit-large-patch14")27 28# The UNet model for generating the latents.29unet = UNet2DConditionModel.from_pretrained("CompVis/stable-diffusion-v1-4", subfolder="unet")30 31# The noise scheduler32scheduler = LMSDiscreteScheduler(beta_start=0.00085, beta_end=0.012, beta_schedule="scaled_linear", num_train_timesteps=1000)33 34# To the GPU we go!35vae = vae.to(torch_device)36text_encoder = text_encoder.to(torch_device)37unet = unet.to(torch_device);38 39 40token_emb_layer = text_encoder.text_model.embeddings.token_embedding41pos_emb_layer = text_encoder.text_model.embeddings.position_embedding42position_ids = text_encoder.text_model.embeddings.position_ids[:, :77]43position_embeddings = pos_emb_layer(position_ids)44