Colab运行ShapeGAN渲染脚本时出现Pygame/ALSA错误求助
解决Colab中ShapeGAN渲染阶段的ALSA音频与Pygame初始化错误
问题场景
在Colab运行ShapeGAN仓库的demo_latent_space.py脚本(用于预训练DeepSDF模型重建隐空间遍历动画)时,执行!python3 demo_latent_space.py后,渲染阶段出现ALSA音频设备错误及Pygame视频系统未初始化报错。
错误日志
Calculating embedding... /usr/local/lib/python3.8/dist-packages/sklearn/manifold/_t_sne.py:780: FutureWarning: The default initialization in TSNE will change from 'random' to 'pca' in 1.2. warnings.warn( /usr/local/lib/python3.8/dist-packages/sklearn/manifold/_t_sne.py:790: FutureWarning: The default learning rate in TSNE will change from 200.0 to 'auto' in 1.2. warnings.warn( Calculating clusters... Calculating trip... 100% 100/100 [00:20<00:00, 4.95it/s] Rendering... ALSA lib confmisc.c:767:(parse_card) cannot find card '0' ALSA lib conf.c:4528:(_snd_config_evaluate) function snd_func_card_driver returned error: No such file or directory ALSA lib confmisc.c:392:(snd_func_concat) error evaluating strings ALSA lib conf.c:4528:(_snd_config_evaluate) function snd_func_concat returned error: No such file or directory ALSA lib confmisc.c:1246:(snd_func_refer) error evaluating name ALSA lib conf.c:4528:(_snd_config_evaluate) function snd_func_refer returned error: No such file or directory ALSA lib conf.c:5007:(snd_config_expand) Evaluate error: No such file or directory ALSA lib pcm.c:2495:(snd_pcm_open_noupdate) Unknown PCM default Traceback (most recent call last): File "demo_latent_space.py", line 160, in <module> viewer = MeshRenderer(size=1080, start_thread=False) File "/content/drive/MyDrive/EnactiveDesign/final-project/grayboxAI/shapegan/rendering/__init__.py", line 91, in __init__ self._initialize_opengl() File "/content/drive/MyDrive/EnactiveDesign/final-project/grayboxAI/shapegan/rendering/__init__.py", line 264, in _initialize_opengl pygame.display.gl_set_attribute(pygame.GL_MULTISAMPLEBUFFERS, 1) pygame.error: video system not initialized
涉事脚本代码(demo_latent_space.py)
from util import device, ensure_directory import scipy.interpolate import numpy as np from rendering import MeshRenderer import torch from tqdm import tqdm import cv2 import random import matplotlib.pyplot as plt from sklearn.manifold import TSNE from matplotlib.offsetbox import Bbox from sklearn.cluster import KMeans SAMPLE_COUNT = 30 # Number of distinct objects to generate and interpolate between TRANSITION_FRAMES = 60 USE_VAE = False SURFACE_LEVEL = 0.011 FRAMES = SAMPLE_COUNT * TRANSITION_FRAMES progress = np.arange(FRAMES, dtype=float) / TRANSITION_FRAMES if USE_VAE: from model.autoencoder import Autoencoder, LATENT_CODE_SIZE vae = Autoencoder() vae.load() vae.eval() print("Calculating latent codes...") from datasets import VoxelDataset from torch.utils.data import DataLoader dataset = VoxelDataset.glob('data/chairs/voxels_32/**.npy') dataloader = DataLoader(dataset, batch_size=1000, num_workers=8) latent_codes = torch.zeros((len(dataset), LATENT_CODE_SIZE)) with torch.no_grad(): position = 0 for batch in tqdm(dataloader): latent_codes[position:position + batch.shape[0], :] = vae.encode(batch.to(device)).detach().cpu() latent_codes = latent_codes.numpy() else: from model.sdf_net import SDFNet, LATENT_CODES_FILENAME latent_codes = torch.load(LATENT_CODES_FILENAME).detach().cpu().numpy() sdf_net = SDFNet() sdf_net.load() sdf_net.eval() from shapenet_metadata import shapenet labels = torch.load('data/labels.to') print("Calculating embedding...") tsne = TSNE(n_components=2) latent_codes_embedded = tsne.fit_transform(latent_codes) print("Calculating clusters...") kmeans = KMeans(n_clusters=SAMPLE_COUNT) indices = np.zeros(SAMPLE_COUNT, dtype=int) kmeans_clusters = kmeans.fit_predict(latent_codes_embedded) for i in range(SAMPLE_COUNT): center = kmeans.cluster_centers_[i, :] cluster_classes = labels[kmeans_clusters == i] cluster_class = np.bincount(cluster_classes).argmax() dist = np.linalg.norm(latent_codes_embedded - center[np.newaxis, :], axis=1) dist[labels != cluster_class] = float('inf') indices[i] = np.argmin(dist) def try_find_shortest_roundtrip(indices): best_order = indices best_distance = None for _ in range(5000): candiate = best_order.copy() a = random.randint(0, SAMPLE_COUNT-1) b = random.randint(0, SAMPLE_COUNT-1) candiate[a] = best_order[b] candiate[b] = best_order[a] dist = np.sum(np.linalg.norm(latent_codes_embedded[candiate, :] - latent_codes_embedded[np.roll(candiate, 1), :], axis=1)).item() if best_distance is None or dist < best_distance: best_distance = dist best_order = candiate return best_order, best_distance def find_shortest_roundtrip(indices): best_order, best_distance = try_find_shortest_roundtrip(indices) for _ in tqdm(range(100)): np.random.shuffle(indices) order, distance = try_find_shortest_roundtrip(indices) if distance < best_distance: best_order = order return best_order print("Calculating trip...") indices = find_shortest_roundtrip(indices) indices = np.concatenate((indices, indices[0][np.newaxis])) SIZE = latent_codes.shape[0] stop_latent_codes = latent_codes[indices, :] colors = np.zeros((labels.shape[0], 3)) for i in range(labels.shape[0]): colors[i, :] = shapenet.get_color(labels[i]) spline = scipy.interpolate.CubicSpline(np.arange(SAMPLE_COUNT + 1), stop_latent_codes, axis=0, bc_type='periodic') frame_latent_codes = spline(progress) color_spline = scipy.interpolate.CubicSpline(np.arange(SAMPLE_COUNT + 1), colors[indices, :], axis=0, bc_type='periodic') frame_colors = color_spline(progress) frame_colors = np.clip(frame_colors, 0, 1) frame_colors = np.zeros((progress.shape[0], 3)) for i in range(SAMPLE_COUNT): frame_colors[i*TRANSITION_FRAMES:(i+1)*TRANSITION_FRAMES, :] = np.linspace(colors[indices[i]], colors[indices[i+1]], num=TRANSITION_FRAMES) embedded_spline = scipy.interpolate.CubicSpline(np.arange(SAMPLE_COUNT + 1), latent_codes_embedded[indices, :], axis=0, bc_type='periodic') frame_latent_codes_embedded = embedded_spline(progress) frame_latent_codes_embedded[0, :] = frame_latent_codes_embedded[-1, :] width, height = 40, 40 PLOT_FILE_NAME = 'tsne.png' ensure_directory('images') margin = 2 range_x = (latent_codes_embedded[:, 0].min() - margin, latent_codes_embedded[:, 0].max() + margin) range_y = (latent_codes_embedded[:, 1].min() - margin, latent_codes_embedded[:, 1].max() + margin) plt.ioff() def create_plot(index, resolution=1080, filename=PLOT_FILE_NAME, dpi=100): frame_color = frame_colors[index, :] frame_color = (frame_color[0], frame_color[1], frame_color[2], 1.0) size_inches = resolution / dpi fig, ax = plt.subplots(1, figsize=(size_inches, size_inches), dpi=dpi) ax.set_position([0, 0, 1, 1]) plt.axis('off') ax.set_xlim(range_x) ax.set_ylim(range_y) ax.plot(frame_latent_codes_embedded[:, 0], frame_latent_codes_embedded[:, 1], c=(0.2, 0.2, 0.2, 1.0), zorder=1, linewidth=2) ax.scatter(latent_codes_embedded[:, 0], latent_codes_embedded[:, 1], c=colors[:SIZE], s = 10, zorder=0) ax.scatter(frame_latent_codes_embedded[index, 0], frame_latent_codes_embedded[index, 1], facecolors=frame_color, s = 200, linewidths=2, edgecolors=(0.1, 0.1, 0.1, 1.0), zorder=2) ax.scatter(latent_codes_embedded[indices, 0], latent_codes_embedded[indices, 1], facecolors=colors[indices, :], s = 140, linewidths=1, edgecolors=(0.1, 0.1, 0.1, 1.0), zorder=3) fig.savefig(filename, bbox_inches=Bbox([[0, 0], [size_inches, size_inches]]), dpi=dpi) plt.close(fig) frame_latent_codes = torch.tensor(frame_latent_codes, dtype=torch.float32, device=device) print("Rendering...") viewer = MeshRenderer(size=1080, start_thread=False) def render_frame(frame_index): viewer.model_color = frame_colors[frame_index, :] with torch.no_grad(): if USE_VAE: viewer.set_voxels(vae.decode(frame_latent_codes[frame_index, :])) else: viewer.set_mesh(sdf_net.get_mesh(frame_latent_codes[frame_index, :], voxel_resolution=128, sphere_only=True, level=SURFACE_LEVEL)) image_mesh = viewer.get_image(flip_red_blue=True) create_plot(frame_index) image_tsne = plt.imread(PLOT_FILE_NAME)[:, :, [2, 1, 0]] * 255 image = np.concatenate((image_mesh, image_tsne), axis=1) cv2.imwrite("images/frame-{:05d}.png".format(frame_index), image) for frame_index in tqdm(range(SAMPLE_COUNT * TRANSITION_FRAMES)): render_frame(frame_index) frame_index += 1 print("\n\nUse this command to create a video:\n") print('ffmpeg -framerate 30 -i images/frame-%05d.png -c:v libx264 -profile:v high -crf 19 -pix_fmt yuv420p video.mp4')
解决方案
1. 安装虚拟显示依赖
Colab是无头环境(无物理显示设备),需安装虚拟显示工具模拟视频输出:
在Colab单元格执行:
!apt-get install -y xvfb x11-utils !pip install pyvirtualdisplay
2. 修改脚本,添加虚拟显示与音频禁用逻辑
在demo_latent_space.py最开头插入以下代码,禁用Pygame音频并启动虚拟显示:
import os # 禁用Pygame音频,规避ALSA设备错误 os.environ['SDL_AUDIODRIVER'] = 'dummy' # 初始化虚拟显示 from pyvirtualdisplay import Display display = Display(visible=0, size=(1080, 1080)) display.start()
3. 验证渲染逻辑(可选)
若问题仍存在,检查shapegan/rendering/__init__.py中MeshRenderer类的_initialize_opengl方法,确保Pygame初始化顺序正确:
- 先调用所有
pygame.display.gl_set_attribute设置属性 - 再执行
pygame.display.set_mode初始化显示
4. 重新运行脚本
修改完成后,再次执行!python3 demo_latent_space.py即可正常渲染帧图像,最后用脚本提示的ffmpeg命令合成视频。
内容的提问来源于stack exchange,提问作者ibib
相关产品推荐
相关产品推荐

