如何用Matplotlib的PdfPages实现每页4个图表的PDF输出?
问题
已有可生成目标图表的Seaborn代码,使用matplotlib.backends.backend_pdf.PdfPages导出PDF时,希望每页展示4个图表。尝试用grouper函数分组处理后,所有图表重复出现在每一页,求正确实现方法或代码问题分析。
原有可正常生成所有图表的代码:
import pandas as pd import numpy as np import seaborn as sns import matplotlib.pyplot as plt from matplotlib.backends.backend_pdf import PdfPages plt.style.use('ggplot') %matplotlib inline data = dict({'Variable_Grouping':['Type_A', 'Type_A', 'Type_A', 'Type_C', 'Type_C', 'Type_C', 'Type_C', 'Type_D', 'Type_D', 'Type_E', 'Type_E', 'Type_E', 'Type_H', 'Type_H'], 'Variable':['a1', 'a2', 'a3', 'c1', 'c2', 'c3', 'c4', 'd1', 'd2', 'e1', 'e2', 'e3', 'h1', 'h2'], 'Count':[5, 3, 8, 4, 3, 9, 5, 3, 8, 5, 3, 8, 5, 3],'Percent':[0.0625, 0.125, 0.4375, 0.0, 0.125, 0.5, 0.02, 0.125, 0.03, 0.0625, 0.05, 0.44, 0.07, 0.023]}) to_plot = pd.DataFrame(data) g = sns.FacetGrid(to_plot, col='Variable_Grouping', col_wrap = 2, sharex=False, sharey = False, height = 5, aspect = 1, margin_titles=True) g=g.map(plt.bar, "Variable","Count").add_legend() for ax, (_, subdata) in zip(g.axes, to_plot.groupby('Variable_Grouping')): ax2=ax.twinx() subdata.plot(x='Variable',y='Percent', ax = ax2, legend=True, color='g', label = 'Percent') ax2.set_ylabel('Percent') ax2.grid(False) for ax in g.axes.flatten(): ax.tick_params(labelbottom=True, labelrotation = 90) g.fig.suptitle('Analysis', fontsize=16, fontweight = 'demibold', y = 1.02) g.fig.subplots_adjust(hspace=0.3, wspace=0.7, right = 0.9) plt.show();
尝试分组导出的问题代码:
def grouper(iterable, n, fillvalue=None): from itertools import zip_longest args = [iter(iterable)] * n return zip_longest(*args, fillvalue=fillvalue) if len(to_plot['Variable_Grouping'].unique()) < 4: N_plots_per_page =len(to_plot['Variable_Grouping'].unique()) elif len(to_plot['Variable_Grouping'].unique()) >= 4: N_plots_per_page = 4 with PdfPages('Analysis.pdf') as pdf: for cols in grouper(to_plot['Variable_Grouping'].unique(), N_plots_per_page): g = sns.FacetGrid(to_plot, col='Variable_Grouping', col_wrap = 2, sharex=False, sharey = False, height = 5, aspect = 1, margin_titles=True) g=g.map(plt.bar, "Variable","Count").add_legend() for ax, (_, subdata) in zip(g.axes, to_plot.groupby('Variable_Grouping')): ax2=ax.twinx() subdata.plot(x='Variable',y='Percent', ax = ax2, legend=True, color='g', label = 'Percent') ax2.set_ylabel('Percent') ax2.grid(False) for ax in g.axes.flatten(): ax.tick_params(labelbottom=True, labelrotation = 90) g.fig.suptitle('Analysis', fontsize=16, fontweight = 'demibold', y = 1.02) g.fig.subplots_adjust(hspace=0.3, wspace=0.7, right = 0.9) pdf.savefig(bbox_inches = 'tight') plt.show() plt.close();
问题分析
核心问题是分组循环完全没用到grouper生成的cols变量,每次创建FacetGrid都传入了完整的to_plot数据,导致每页都绘制所有Variable_Grouping的图表,自然出现重复。
另外还有两个小问题:
- grouper生成的
cols可能包含fillvalue(比如最后一页不足4个分组时),需要过滤掉空值 - 循环中遍历
to_plot.groupby('Variable_Grouping')时,没有和当前页的分组对应,会导致双轴数据匹配错误
修正后的代码
import pandas as pd import numpy as np import seaborn as sns import matplotlib.pyplot as plt from matplotlib.backends.backend_pdf import PdfPages from itertools import zip_longest plt.style.use('ggplot') data = dict({'Variable_Grouping':['Type_A', 'Type_A', 'Type_A', 'Type_C', 'Type_C', 'Type_C', 'Type_C', 'Type_D', 'Type_D', 'Type_E', 'Type_E', 'Type_E', 'Type_H', 'Type_H'], 'Variable':['a1', 'a2', 'a3', 'c1', 'c2', 'c3', 'c4', 'd1', 'd2', 'e1', 'e2', 'e3', 'h1', 'h2'], 'Count':[5, 3, 8, 4, 3, 9, 5, 3, 8, 5, 3, 8, 5, 3],'Percent':[0.0625, 0.125, 0.4375, 0.0, 0.125, 0.5, 0.02, 0.125, 0.03, 0.0625, 0.05, 0.44, 0.07, 0.023]}) to_plot = pd.DataFrame(data) def grouper(iterable, n, fillvalue=None): args = [iter(iterable)] * n return zip_longest(*args, fillvalue=fillvalue) unique_groups = to_plot['Variable_Grouping'].unique() N_plots_per_page = min(4, len(unique_groups)) with PdfPages('Analysis.pdf') as pdf: for group_batch in grouper(unique_groups, N_plots_per_page): # 过滤掉grouper填充的空值 current_groups = [g for g in group_batch if g is not None] # 筛选当前页要绘制的数据 page_data = to_plot[to_plot['Variable_Grouping'].isin(current_groups)] # 创建仅包含当前分组的FacetGrid g = sns.FacetGrid(page_data, col='Variable_Grouping', col_wrap=2, sharex=False, sharey=False, height=5, aspect=1, margin_titles=True) g = g.map(plt.bar, "Variable", "Count").add_legend() # 遍历当前页的分组和对应的轴,匹配双轴数据 grouped_data = page_data.groupby('Variable_Grouping') for ax, (group_name, subdata) in zip(g.axes, grouped_data): ax2 = ax.twinx() subdata.plot(x='Variable', y='Percent', ax=ax2, legend=True, color='g', label='Percent') ax2.set_ylabel('Percent') ax2.grid(False) # 调整轴标签 for ax in g.axes.flatten(): ax.tick_params(labelbottom=True, labelrotation=90) # 调整图表布局和标题 g.fig.suptitle('Analysis', fontsize=16, fontweight='demibold', y=1.02) g.fig.subplots_adjust(hspace=0.3, wspace=0.7, right=0.9) pdf.savefig(bbox_inches='tight') plt.close()
关键修改点
- 每次循环筛选当前页对应的
current_groups,生成仅包含这些分组的page_data - 创建
FacetGrid时传入page_data,而非完整数据集 - 遍历
page_data.groupby('Variable_Grouping')确保双轴数据和当前页的图表一一对应 - 过滤grouper生成的空值,避免无效图表
- 简化
N_plots_per_page的判断逻辑,用min(4, len(unique_groups))更简洁
内容的提问来源于stack exchange,提问作者Sid
相关产品推荐
相关产品推荐

