使用Hugging Face时PyTorch报GPU内存不足错误求助
问题描述
使用Hugging Face Diffusers库的StableDiffusionControlNetPipeline结合ControlNet生成图像时,当设置num_images_per_prompt(对应代码中的image_count)大于1时,触发PyTorch CUDA显存不足错误,提示尝试分配1.43 GiB失败。
实现代码:
from diffusers import StableDiffusionControlNetPipeline, ControlNetModel, UniPCMultistepScheduler from diffusers.utils import load_image import cv2 from PIL import Image import numpy as np import torch import urllib.request import requests import json import os from django.core.cache import cache import uuid class GenerateImages(): def __init__(self): self.device = 'cuda:0' self.controlnet = ControlNetModel.from_pretrained("lllyasviel/sd-controlnet-canny", torch_dtype=torch.float16).to(self.device) self.pipe = StableDiffusionControlNetPipeline.from_pretrained("runwayml/stable-diffusion-v1-5", controlnet=self.controlnet, torch_dtype=torch.float16, use_safetensors=True).to(self.device) self.pipe.scheduler = UniPCMultistepScheduler.from_config(self.pipe.scheduler.config) self.pipe.enable_model_cpu_offload() self.pipe.enable_xformers_memory_efficient_attention() def generate_images(self, canny_image,org_image_path, prompt,image_count, user_folder='generated'): org_image=Image.open(org_image_path) output_list = [] if(torch.cuda.is_available()): generator = torch.Generator(device=self.device).manual_seed(0) else: generator = torch.Generator().manual_seed(0) prompt = [prompt] output = self.pipe( prompt, canny_image, negative_prompt=['monochrome, lowres, bad anatomy,unrealistic,composite , worst quality, low quality'], generator=generator, num_inference_steps=20, num_images_per_prompt=image_count, ).images image_list = [] for x in output: temp_image = x temp_image.paste(org_image, (0, 0), org_image) image_list.append(temp_image) processed_folder = os.path.join('media', user_folder) if not os.path.exists(processed_folder): os.makedirs(processed_folder, exist_ok=True) processed_images = image_list processed_image_paths=[] for i, processed_image in enumerate(processed_images): unique_filename = str(uuid.uuid4()) processed_image_path = os.path.join(processed_folder, f'processed_image_{unique_filename}_{i + 1}.png') processed_image.save(processed_image_path) processed_image_paths.append(processed_image_path) torch.cuda.empty_cache() print(image_list,processed_image_paths) return image_list,processed_image_paths
报错堆栈:
File "/usr/local/lib/python3.10/dist-packages/torch/nn/modules/module.py", line 1532, in _wrapped_call_impl return self._call_impl(*args, **kwargs) File "/usr/local/lib/python3.10/dist-packages/torch/nn/modules/module.py", line 1541, in _call_impl return forward_call(*args, **kwargs) File "/usr/local/lib/python3.10/dist-packages/accelerate/hooks.py", line 166, in new_forward output = module._old_forward(*args, **kwargs) File "/usr/local/lib/python3.10/dist-packages/diffusers/models/controlnet.py", line 797, in forward controlnet_cond = self.controlnet_cond_embedding(controlnet_cond) File "/usr/local/lib/python3.10/dist-packages/torch/nn/modules/module.py", line 1532, in _wrapped_call_impl return self._call_impl(*args, **kwargs) File "/usr/local/lib/python3.10/dist-packages/torch/nn/modules/module.py", line 1541, in _call_impl return forward_call(*args, **kwargs) File "/usr/local/lib/python3.10/dist-packages/diffusers/models/controlnet.py", line 100, in forward embedding = F.silu(embedding) File "/usr/local/lib/python3.10/dist-packages/torch/nn/functional.py", line 2102, in silu return torch._C._nn.silu(input) torch.cuda.OutOfMemoryError: CUDA out of memory. Tried to allocate 1.43 GiB. GPU
解决思路
分批次生成图像:放弃一次性生成多张的方式,循环调用生成方法每次生成1张,避免显存一次性加载多批次张量。修改
generate_images方法中的生成逻辑:# 替换原pipe调用代码块 image_list = [] for _ in range(image_count): output = self.pipe( prompt, canny_image, negative_prompt=['monochrome, lowres, bad anatomy,unrealistic,composite , worst quality, low quality'], generator=generator, num_inference_steps=20, num_images_per_prompt=1, ).images[0] temp_image = output temp_image.paste(org_image, (0, 0), org_image) image_list.append(temp_image) torch.cuda.empty_cache() # 每次生成后释放临时显存降低图像分辨率:显存占用与图像分辨率平方成正比,将输入的canny图和原图缩放到更小尺寸(比如从512x512降到384x384),生成后再按需放大。示例:
# 在generate_images方法开头添加分辨率处理 canny_image = canny_image.resize((384, 384), Image.LANCZOS) org_image = org_image.resize((384, 384), Image.LANCZOS)启用VAE切片:在
__init__方法中添加VAE切片功能,让VAE分块处理图像,减少显存占用:def __init__(self): # 原代码... self.pipe.enable_xformers_memory_efficient_attention() self.pipe.enable_vae_slicing() # 新增该行使用模型量化加载:借助
bitsandbytes库以8-bit/4-bit精度加载模型,大幅降低显存占用。修改模型加载代码:self.controlnet = ControlNetModel.from_pretrained( "lllyasviel/sd-controlnet-canny", torch_dtype=torch.float16, load_in_8bit=True, device_map="auto" ) self.pipe = StableDiffusionControlNetPipeline.from_pretrained( "runwayml/stable-diffusion-v1-5", controlnet=self.controlnet, torch_dtype=torch.float16, use_safetensors=True, load_in_8bit=True, device_map="auto" )注意:需要先安装
bitsandbytes库。优化内存占用:生成图像后立即保存并从内存中移除临时对象,避免大量图像数据堆积在内存中。
内容的提问来源于stack exchange,提问作者sreerag m
相关产品推荐
相关产品推荐

