modelscope/DiffSynth-Painter
14
1import torch2import numpy as np3from PIL import Image4from .base import VideoProcessor5 6 7class RIFESmoother(VideoProcessor):8 def __init__(self, model, device="cuda", scale=1.0, batch_size=4, interpolate=True):9 self.model = model10 self.device = device11 12 # IFNet only does not support float1613 self.torch_dtype = torch.float3214 15 # Other parameters16 self.scale = scale17 self.batch_size = batch_size18 self.interpolate = interpolate19 20 @staticmethod21 def from_model_manager(model_manager, **kwargs):22 return RIFESmoother(model_manager.RIFE, device=model_manager.device, **kwargs)23 24 def process_image(self, image):25 width, height = image.size26 if width % 32 != 0 or height % 32 != 0:27 width = (width + 31) // 3228 height = (height + 31) // 3229 image = image.resize((width, height))30 image = torch.Tensor(np.array(image, dtype=np.float32)[:, :, [2,1,0]] / 255).permute(2, 0, 1)31 return image32 33 def process_images(self, images):34 images = [self.process_image(image) for image in images]35 images = torch.stack(images)36 return images37 38 def decode_images(self, images):39 images = (images[:, [2,1,0]].permute(0, 2, 3, 1) * 255).clip(0, 255).numpy().astype(np.uint8)40 images = [Image.fromarray(image) for image in images]41 return images42 43 def process_tensors(self, input_tensor, scale=1.0, batch_size=4):44 output_tensor = []45 for batch_id in range(0, input_tensor.shape[0], batch_size):46 batch_id_ = min(batch_id + batch_size, input_tensor.shape[0])47 batch_input_tensor = input_tensor[batch_id: batch_id_]48 batch_input_tensor = batch_input_tensor.to(device=self.device, dtype=self.torch_dtype)49 flow, mask, merged = self.model(batch_input_tensor, [4/scale, 2/scale, 1/scale])50 output_tensor.append(merged[2].cpu())51 output_tensor = torch.concat(output_tensor, dim=0)52 return output_tensor53 54 @torch.no_grad()55 def __call__(self, rendered_frames, **kwargs):56 # Preprocess57 processed_images = self.process_images(rendered_frames)58 59 # Input60 input_tensor = torch.cat((processed_images[:-2], processed_images[2:]), dim=1)61 62 # Interpolate63 output_tensor = self.process_tensors(input_tensor, scale=self.scale, batch_size=self.batch_size)64 65 if self.interpolate:66 # Blend67 input_tensor = torch.cat((processed_images[1:-1], output_tensor), dim=1)68 output_tensor = self.process_tensors(input_tensor, scale=self.scale, batch_size=self.batch_size)69 processed_images[1:-1] = output_tensor70 else:71 processed_images[1:-1] = (processed_images[1:-1] + output_tensor) / 272 73 # To images74 output_images = self.decode_images(processed_images)75 if output_images[0].size != rendered_frames[0].size:76 output_images = [image.resize(rendered_frames[0].size) for image in output_images]77 return output_images78 