You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用QwenImageEditPlus与GGUF权重时设备不匹配问题的解决方法

解决QwenImageEditPlus设备不匹配的RuntimeError问题

问题背景

尝试运行QwenImageEditPlus失败后,改用基于GGUF量化模型的方案,但在Kaggle双T4 GPU环境下,出现张量索引时的设备不匹配错误。

运行环境

  • 平台:Kaggle
  • GPU:T4×2(单卡15GB)
  • 内存:30GB

运行代码

import spaces
import gradio as gr
import torch
import math
from PIL import Image
from diffusers import QwenImageEditPlusPipeline, FlowMatchEulerDiscreteScheduler
from diffusers import QwenImageTransformer2DModel, GGUFQuantizationConfig
from transformers import AutoModelForCausalLM, AutoTokenizer

# Load pipeline with optimized scheduler at startup
scheduler_config = {
"base_image_seq_len": 256,
"base_shift": math.log(3),
"invert_sigmas": False,
"max_image_seq_len": 8192,
"max_shift": math.log(3),
"num_train_timesteps": 1000,
"shift": 1.0,
"shift_terminal": None,
"stochastic_sampling": False,
"time_shift_type": "exponential",
"use_beta_sigmas": False,
"use_dynamic_shifting": True,
"use_exponential_sigmas": False,
"use_karras_sigmas": False,
}
scheduler = FlowMatchEulerDiscreteScheduler.from_config(scheduler_config)

model_path = "https://huggingface.co/calcuis/qwen-image-edit-plus-gguf/blob/main/qwen-image-edit-plus-iq2_s.gguf" 

transformer = QwenImageTransformer2DModel.from_single_file(
model_path,
quantization_config=GGUFQuantizationConfig(compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
config="callgg/image-edit-decoder",
subfolder="transformer"
)

pipeline = QwenImageEditPlusPipeline.from_pretrained("callgg/image-edit-decoder", 
                                                 transformer=transformer, 
                                                 scheduler=scheduler,
                                                 torch_dtype=torch.bfloat16)

print("Pipeline Loaded")

onload_device0 = torch.device("cuda:0")
onload_device1= torch.device("cuda:1")
offload_device=torch.device("cpu")

from diffusers.hooks import apply_group_offloading
# Use the apply_group_offloading method for other model components
apply_group_offloading(pipeline.text_encoder, onload_device=onload_device0, offload_type="leaf_level")
apply_group_offloading(pipeline.transformer, onload_device=onload_device1, offload_type="leaf_level")
apply_group_offloading(pipeline.vae, onload_device=onload_device1, offload_type="leaf_level")

image1 = "/kaggle/input/image1.png"
image2 = "/kaggle/input/image2.png"
prompt = "Do something ..."

if image1 is None or image2 is None:
    gr.Warning("Please upload both images")
    return None

# Convert to PIL if needed
if not isinstance(image1, Image.Image):
    image1 = Image.fromarray(image1)
if not isinstance(image2, Image.Image):
    image2 = Image.fromarray(image2)


inputs = {
    "image": [image1, image2],
    "prompt": prompt,
    "generator": torch.manual_seed(seed),
    "true_cfg_scale": true_cfg_scale,
    "negative_prompt": negative_prompt,
    "num_inference_steps": num_steps,
    "guidance_scale": guidance_scale,
    "num_images_per_prompt": 1,
}

pipeline.enable_attention_slicing()
pipeline.vae.enable_tiling()

with torch.inference_mode():
    output = pipeline(**inputs)
    return output.images[0]

报错信息

File "/usr/local/lib/python3.11/dist-packages/diffusers/pipelines/qwenimage/pipeline_qwenimage_edit_plus.py", line 700, in __call__
    prompt_embeds, prompt_embeds_mask = self.encode_prompt(
                                    ^^^^^^^^^^^^^^^^^^^
  File "/usr/local/lib/python3.11/dist-packages/diffusers/pipelines/qwenimage/pipeline_qwenimage_edit_plus.py", line 318, in encode_prompt
    prompt_embeds, prompt_embeds_mask = self._get_qwen_prompt_embeds(prompt, image, device)
                                    ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/usr/local/lib/python3.11/dist-packages/diffusers/pipelines/qwenimage/pipeline_qwenimage_edit_plus.py", line 271, in _get_qwen_prompt_embeds
    split_hidden_states = self._extract_masked_hidden(hidden_states, model_inputs.attention_mask)
                      ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
  File "/usr/local/lib/python3.11/dist-packages/diffusers/pipelines/qwenimage/pipeline_qwenimage_edit_plus.py", line 224, in _extract_masked_hidden
    selected = hidden_states[bool_mask]
           ~~~~~~~~~~~~~^^^^^^^^^^^
RuntimeError: indices should be either on cpu or on the same device as the indexed tensor (cuda:1)

解决方案

报错根源是手动分组卸载时,bool_mask张量在CPU,而hidden_states在cuda:1,导致跨设备索引操作失败。以下是几种可行的修复方案:

方案1:使用官方自动设备分配与卸载

放弃手动分组卸载,改用Diffusers官方的自动设备管理,稳定性更高:

# 替换原有的apply_group_offloading代码段
# 加载管道时指定device_map,让框架自动分配组件到GPU
pipeline = QwenImageEditPlusPipeline.from_pretrained(
    "callgg/image-edit-decoder", 
    transformer=transformer, 
    scheduler=scheduler,
    torch_dtype=torch.bfloat16,
    device_map="auto"
)

# 启用CPU卸载(单卡内存不足时使用)
pipeline.enable_model_cpu_offload()

# 保留原有的优化设置
pipeline.enable_attention_slicing()
pipeline.vae.enable_tiling()

方案2:猴子补丁修复管道内部方法

直接修改管道的_extract_masked_hidden方法,确保索引张量与目标张量在同一设备:

# 在加载管道后添加这段代码
from diffusers.pipelines.qwenimage.pipeline_qwenimage_edit_plus import QwenImageEditPlusPipeline

def patched_extract_masked_hidden(self, hidden_states, attention_mask):
    # 将attention_mask移到hidden_states所在设备
    bool_mask = attention_mask.to(hidden_states.device)
    selected = hidden_states[bool_mask]
    return selected

# 替换原方法
QwenImageEditPlusPipeline._extract_masked_hidden = patched_extract_masked_hidden

方案3:统一输入张量设备

确保所有输入相关的张量(如generator)与管道核心组件在同一设备:

# 在构建inputs字典后添加
device = pipeline.transformer.device  # 获取transformer所在设备
inputs["generator"] = torch.manual_seed(seed).to(device)

方案4:优先单卡运行

双卡手动卸载容易出现设备同步问题,优先尝试单卡运行:

# 加载管道时指定单卡
pipeline = QwenImageEditPlusPipeline.from_pretrained(
    "callgg/image-edit-decoder", 
    transformer=transformer, 
    scheduler=scheduler,
    torch_dtype=torch.bfloat16,
    device_map="cuda:0"
)

# 启用必要优化
pipeline.enable_attention_slicing()
pipeline.vae.enable_tiling()
# 若单卡内存不足,启用CPU卸载
pipeline.enable_model_cpu_offload()

内容的提问来源于stack exchange,提问作者Siladittya

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.12 07:55:53