使用Python ReportLab从DataFrame生成PDF的三个问题求助
解决ReportLab生成PDF报表的三个需求
针对你提到的三个功能需求,下面是具体的实现方案和修改后的完整代码:
1. 每页显示表头
ReportLab默认的Table不支持跨页重复表头,需要自定义Flowable类重写split方法,实现分页时自动插入表头行。
2. 指定列宽
创建Table对象时,通过colWidths参数直接设置各列宽度,支持固定数值或按页面比例分配两种方式。
3. 单元格文本自动换行并自适应高度
将单元格内容转换为Paragraph对象,配合自定义ParagraphStyle开启自动换行,Table会根据内容自动调整行高。
修改后的完整代码
import pandas as pd from reportlab.lib.pagesizes import letter from reportlab.platypus import SimpleDocTemplate, Table, TableStyle, Paragraph, Flowable from reportlab.lib.styles import getSampleStyleSheet, ParagraphStyle from reportlab.lib.units import inch # 自定义带重复表头的表格类 class TableWithHeader(Flowable): def __init__(self, table_data, col_widths, table_style): self.table_data = table_data self.col_widths = col_widths self.table_style = table_style self.header = table_data[0] self.data_rows = table_data[1:] self.page_height = letter[1] - 2*inch # 减去上下边距 def wrap(self, avail_width, avail_height): # 计算单一行的高度 sample_table = Table([self.header] + [self.data_rows[0]], colWidths=self.col_widths) sample_table.setStyle(self.table_style) _, row_height = sample_table.wrap(avail_width, avail_height) self.row_height = row_height return avail_width, len(self.data_rows)*self.row_height def split(self, avail_width, avail_height): # 计算每页能容纳的数据行数 max_rows_per_page = int((self.page_height - self.row_height) // self.row_height) chunks = [] for i in range(0, len(self.data_rows), max_rows_per_page): chunk_data = [self.header] + self.data_rows[i:i+max_rows_per_page] chunk_table = Table(chunk_data, colWidths=self.col_widths) chunk_table.setStyle(self.table_style) chunks.append(chunk_table) return chunks # Path to the Excel file file_path = 'file.xlsx' # Load the second sheet into a data frame, skipping two rows df = pd.read_excel(file_path, sheet_name='mainsheet', skiprows=2) # Columns used for grouping group_columns = ['Country', 'State', 'City'] grouped_df = df.groupby(group_columns) # Create the PDF document doc = SimpleDocTemplate("output.pdf", pagesize=letter, leftMargin=0.5*inch, rightMargin=0.5*inch, topMargin=0.5*inch, bottomMargin=0.5*inch) # Create a list to hold the PDF content elements = [] # 定义单元格文本样式(支持自动换行) cell_style = ParagraphStyle( name='CellStyle', fontName='Helvetica', fontSize=10, leading=12, # 行间距 wordWrap='CJK' # 自动换行,兼容中英文 ) # Define the table style table_style = TableStyle([ ('BACKGROUND', (0, 0), (-1, 0), 'gray'), ('TEXTCOLOR', (0, 0), (-1, 0), 'white'), ('ALIGN', (0, 0), (-1, 0), 'CENTER'), ('FONTNAME', (0, 0), (-1, 0), 'Helvetica-Bold'), ('FONTSIZE', (0, 0), (-1, 0), 12), ('BOTTOMPADDING', (0, 0), (-1, 0), 12), ('BACKGROUND', (0, 1), (-1, -1), 'white'), ('GRID', (0, 0), (-1, -1), 1, 'gray'), ('VALIGN', (0, 0), (-1, -1), 'TOP'), # 单元格内容顶部对齐 ]) # 指定列宽(示例:按页面宽度比例分配,总宽度减去边距) total_width = letter[0] - doc.leftMargin - doc.rightMargin col_widths = [total_width*0.2, total_width*0.2, total_width*0.2, total_width*0.4] # 四列按2:2:2:4分配 # Iterate over the grouped dataframe for group_key, group_df in grouped_df: # Add the group information as a header group_info = "Group: " + " / ".join(str(val) for val in group_key) header_paragraph = Paragraph(group_info, getSampleStyleSheet()['Heading2']) elements.append(header_paragraph) # 将DataFrame数据转换为包含Paragraph的列表(支持自动换行) header = [Paragraph(col, getSampleStyleSheet()['Heading4']) for col in group_df.columns.tolist()] data_rows = [] for row in group_df.values.tolist(): data_row = [Paragraph(str(cell), cell_style) for cell in row] data_rows.append(data_row) table_data = [header] + data_rows # 创建带重复表头的表格 table = TableWithHeader(table_data, col_widths, table_style) elements.append(table) # Add a page break after each group elements.append(Paragraph("<br/><br/>", getSampleStyleSheet()['Normal'])) # Build the PDF document doc.build(elements)
关键修改说明
- 每页重复表头:通过
TableWithHeader类的split方法,在分页时自动截取数据块并添加表头。 - 指定列宽:
col_widths参数可设置固定值(如[1*inch, 1*inch])或比例值,适配不同页面需求。 - 自动换行与自适应高度:所有单元格内容转为
Paragraph对象,开启wordWrap属性实现自动换行,Table会根据内容高度自动调整行高。
内容的提问来源于stack exchange,提问作者Prabhat Sharma
相关产品推荐
相关产品推荐

