搭载MPS的MacOS运行Hugging Face Stable Diffusion模型报错求助
在搭载MPS的macOS上运行SDXL模型的错误解决建议
问题背景
在搭载MPS的macOS系统中运行基于Hugging Face的Stable Diffusion XL模型代码时,所有图形相关模型均触发相同维度不匹配错误,无法生成预期的PIL格式图像。
运行代码
import gradio as gr import torch import numpy as np import modin.pandas as pd from PIL import Image from diffusers import DiffusionPipeline import os os.environ["PYTORCH_MPS_HIGH_WATERMARK_RATIO"] = "0.0" device = 'cuda' if torch.cuda.is_available() else 'mps' if torch.cuda.is_available(): PYTORCH_CUDA_ALLOC_CONF = {'max_split_size_mb': 8000} torch.cuda.max_memory_allocated(device=device) torch.cuda.empty_cache() pipe = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-base-1.0", torch_dtype=torch.float16, variant="fp16", use_safetensors=True) pipe.enable_xformers_memory_efficient_attention() pipe = pipe.to(device) torch.cuda.empty_cache() refiner = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-refiner-1.0", use_safetensors=True, torch_dtype=torch.float16, variant="fp16") refiner.enable_xformers_memory_efficient_attention() refiner = refiner.to(device) torch.cuda.empty_cache() upscaler = DiffusionPipeline.from_pretrained("stabilityai/sd-x2-latent-upscaler", torch_dtype=torch.float16, use_safetensors=True) upscaler.enable_xformers_memory_efficient_attention() upscaler = upscaler.to(device) torch.cuda.empty_cache() else: pipe = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-base-1.0", use_safetensors=True) pipe = pipe.to(device) pipe.unet = torch.compile(pipe.unet, mode="reduce-overhead", fullgraph=True) refiner = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-refiner-1.0", use_safetensors=True) refiner = refiner.to(device) refiner.unet = torch.compile(refiner.unet, mode="reduce-overhead", fullgraph=True) n_steps = 40 high_noise_frac = 0.8 pipe.enable_attention_slicing() PYTORCH_MPS_HIGH_WATERMARK_RATIO=0.0 def genie(prompt, negative_prompt, height, width, scale, steps, seed, upscaling, prompt_2, negative_prompt_2): generator = torch.Generator(device=device).manual_seed(seed) int_image = pipe(prompt, prompt_2=prompt_2, negative_prompt=negative_prompt, negative_prompt_2=negative_prompt_2, num_inference_steps=steps, height=height, width=width, guidance_scale=scale, num_images_per_prompt=1, generator=generator, output_type="latent").images if upscaling == 'Yes': image = \ refiner(prompt=prompt, prompt_2=prompt_2, negative_prompt=negative_prompt, negative_prompt_2=negative_prompt_2, image=int_image).images[0] upscaled = upscaler(prompt=prompt, negative_prompt=negative_prompt, image=image, num_inference_steps=5, guidance_scale=0).images[0] torch.cuda.empty_cache() return (image, upscaled) else: image = \ refiner(prompt=prompt, prompt_2=prompt_2, negative_prompt=negative_prompt, negative_prompt_2=negative_prompt_2, image=int_image).images[0] torch.cuda.empty_cache() return (image, image) gr.Interface(fn=genie, inputs=[gr.Textbox( label='What you want the AI to generate. 77 Token Limit. A Token is Any Word, Number, Symbol, or Punctuation. Everything Over 77 Will Be Truncated!'), gr.Textbox(label='What you Do Not want the AI to generate. 77 Token Limit'), gr.Slider(512, 1024, 768, step=128, label='Height'), gr.Slider(512, 1024, 768, step=128, label='Width'), gr.Slider(1, 15, 10, step=.25, label='Guidance Scale: How Closely the AI follows the Prompt'), gr.Slider(25, maximum=100, value=50, step=25, label='Number of Iterations'), gr.Slider(minimum=1, step=1, maximum=999999999999999999, randomize=True, label='Seed'), gr.Radio(['Yes', 'No'], value='No', label='Upscale?'), gr.Textbox(label='Embedded Prompt'), gr.Textbox(label='Embedded Negative Prompt')], outputs=['image', 'image'], title="Stable Diffusion XL 1.0 GPU", description="SDXL 1.0 GPU. <br><br><b>WARNING: Capable of producing NSFW (Softcore) images.</b>", article="If You Enjoyed this Demo and would like to Donate, you can send to any of these Wallets. <br>BTC: bc1qzdm9j73mj8ucwwtsjx4x4ylyfvr6kp7svzjn84 <br>3LWRoKYx6bCLnUrKEdnPo3FCSPQUSFDjFP <br>DOGE: DK6LRc4gfefdCTRk9xPD239N31jh9GjKez <br>SHIB (BEP20): 0xbE8f2f3B71DFEB84E5F7E3aae1909d60658aB891 <br>ETH: 0xbE8f2f3B71DFEB84E5F7E3aae1909d60658aB891 <br>Code Monkey: Manjushri").launch( debug=True, max_threads=80)
触发错误
step_index = (self.timesteps == timestep).nonzero().item() ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ RuntimeError: Expected dst.dim() >= src.dim() to be true, but got false. (Could this error message be improved? If so, please report an enhancement request to PyTorch.)
解决建议
禁用torch.compile:MPS后端对
torch.compile的兼容性不足,这是引发维度错误的核心原因之一。修改MPS分支代码,移除对unet的编译逻辑:else: pipe = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-base-1.0", use_safetensors=True) pipe = pipe.to(device) # 移除该行:pipe.unet = torch.compile(pipe.unet, mode="reduce-overhead", fullgraph=True) refiner = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-refiner-1.0", use_safetensors=True) refiner = refiner.to(device) # 移除该行:refiner.unet = torch.compile(refiner.unet, mode="reduce-overhead", fullgraph=True)指定float32数据类型:MPS对float16的支持有限,强制使用float32可避免维度转换异常:
else: pipe = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-base-1.0", use_safetensors=True, torch_dtype=torch.float32) pipe = pipe.to(device) refiner = DiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-xl-refiner-1.0", use_safetensors=True, torch_dtype=torch.float32) refiner = refiner.to(device)统一timesteps设备:手动确保timesteps与模型在同一设备上,避免维度不匹配:
在genie函数生成int_image前添加:pipe.timesteps = pipe.timesteps.to(device) refiner.timesteps = refiner.timesteps.to(device)更新依赖库:确保torch、diffusers、transformers为最新兼容版本,修复已知MPS适配问题:
pip install --upgrade torch diffusers transformers accelerate关闭attention slicing(可选):部分场景下
enable_attention_slicing会与MPS后端冲突,尝试注释掉该行:# pipe.enable_attention_slicing()
内容的提问来源于stack exchange,提问作者B.AL
相关产品推荐
相关产品推荐

