如何将4张发票PDF合并为单页A4尺寸PDF?
问题
我用Python的reportlab库开发了一个程序,以Excel表格为输入,为表格的每一行生成一份PDF发票。代码中还添加了两张图片、横竖线条等元素来构建结构化的发票样式,所有线条和文本的坐标均为硬编码——当时不清楚更优方案,且产品列表至少2年不变。
目前程序运行正常,每行对应一份PDF发票,但现在需要将4张发票合并到单页A4 PDF中以节省纸张,仅拥有A4打印机。作为新手毫无头绪,现有两个思路:
- 编写另一个函数直接生成包含4行数据的PDF,但现有代码硬编码坐标,调整起来非常繁琐;
- 是否有Python模块可以直接合并PDF?
现有代码:
import pandas as pd import xlsxwriter from reportlab.pdfgen import canvas from reportlab.lib.colors import HexColor from reportlab.lib.pagesizes import letter from reportlab.lib.units import inch from reportlab.lib.utils import ImageReader import openpyxl import datetime import os from mergePDF import merge_pdfs #df = pd.read_excel('Invoice.xlsx', usecols=['Sr.No','Cust_name','Quantity','Rate','Total'],header=0) #print(df) def create_pdf(name, cow, crate, cm_total, buffalo, brate, bm_total,other, pending_bill, amount,file_path,image_path, qr_code_path,month): c = canvas.Canvas(file_path) c.setFont('Helvetica-Bold', 20) c.drawString(50,750,'Invoice ') c.setFont('Helvetica', 15) c.drawString(50,730,'Contact : XYZ ') c.drawString(50,710,'Bill for month -'+ month) #add date time today = datetime.datetime.today().strftime('%d-%m-%Y') c.drawString(450, 780, "Date: " + today) #Logo 1 big left #image = ImageReader(image_path) #c.drawImage(image,50,400,width=4*inch,height=4*inch) # Logo 2 small right side image = ImageReader(image_path) c.drawImage(image, 450, 700, width=1 * inch, height=1 * inch) # Adding QR Code qrcode = ImageReader(qr_code_path) c.drawImage(qrcode,350,130, width=3 * inch,height=3 * inch) #add a line seperator c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 700, 550, 700) #Vertical line c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 700, 45, 360) c.line(550, 700, 550, 360) c.line(150, 670, 150, 520) c.line(250, 670, 250, 520) c.line(350, 670, 350, 520) c.setFont('Helvetica', 15) c.drawString(50,680,'Customer Name:') c.drawString(250, 680, name) c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 670, 550, 670) #First line - Product , Litre , Rate , Subtotal c.drawString(50,630,'Product') c.drawString(200,630,'Litre') c.drawString(300,630, 'Rate') c.drawString(450,630, 'Sub Total') c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 620, 550, 620) #Cow milk c.drawString(50, 580, 'Cow Milk') c.drawString(200, 580, str(cow)) c.drawString(300, 580, str(crate)) c.drawString(450, 580, str(cm_total)) c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 570, 550, 570) #Buffalo milk c.drawString(50, 530, 'Buffalo Milk') c.drawString(200, 530, str(buffalo)) c.drawString(300, 530, str(brate)) c.drawString(450, 530, str(bm_total)) c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 520, 550, 520) # Other Items c.drawString(50, 470, 'Other Items') c.drawString(450, 470, str(other)) c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 460, 550, 460) # Previous pending c.drawString(50, 420, 'Pending Bill') c.drawString(450, 420, str(pending_bill)) #c.drawString(350,650,'Total Quantity: '+ str(qty)) #c.drawString(350,600,'Rate per litre: ' + str(rate)) # add a line seperator c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 410, 550, 410) c.setFont('Helvetica-Bold', 16) c.drawString(50, 370, 'Total amount: ' ) c.drawString(450, 370, str(amount)) c.setStrokeColor(HexColor('#000000')) c.setLineWidth(2) c.line(45, 360, 550, 360) # Footer c.setFont('Helvetica', 15) c.drawString(50, 300, " Bank Name :") c.drawString(50, 280, " Account name : ") c.drawString(50, 260, " Account number : ") c.drawString(50, 240, " IFSC code : ") c.drawString(50, 220, " GPAY No : ") c.save() def generate_bills(excel_path,pdf_output_path,image_path, qr_code_path): df = pd.read_excel(excel_path) df = df.fillna(0) # This ensures if value is null then 0 is added to avoid printing NAN/NONE for index ,row in df.iterrows(): name = row['Cust_name'] cow = row['Cow'] crate = row['C_rate'] cm_total = row['CM_total'] buffalo = row['Buffalo'] brate = row['B_rate'] bm_total = row['BM_total'] other = row['Other'] pending_bill = row['Previous_pending'] amount = row['Total'] month = row['Month'] pdf_file_name = f"{name}.pdf" pdf_file_path = f"{pdf_output_path}/{pdf_file_name}" create_pdf(name, cow, crate, cm_total, buffalo, brate,bm_total,other, pending_bill, amount, pdf_file_path, image_path, qr_code_path,month ) def main(): # Get the absolute path of the current working directory current_dir = os.path.abspath(os.getcwd()) # Specify the name of the subdirectory for output files output_subdir = "Output" # Combine the current directory path with the output subdirectory path pdf_output_path = os.path.join(current_dir, output_subdir) pdf_print_path = os.path.join(current_dir,output_print) # Check if the output directory exists, and create it if it does not if not os.path.exists(pdf_output_path): os.makedirs(pdf_output_path) # Specify the path of the input Excel file and image file excel_path = os.path.join(current_dir, "Invoice.xlsx") image_path = os.path.join(current_dir, "logo.jpg") qr_code_path = os.path.join(current_dir, "QR_code.jpg") generate_bills(excel_path, pdf_output_path, image_path, qr_code_path) if __name__ == '__main__': main()
解决方案
优先选择第二个思路,直接合并PDF无需修改原有硬编码的绘制逻辑,效率更高。推荐使用PyPDF2库实现,步骤如下:
1. 安装PyPDF2
pip install PyPDF2
2. 添加合并函数
在现有代码中新增一个函数,将生成的单个发票PDF按每4张一组,以2x2布局合并到单张A4页面:
from PyPDF2 import PdfReader, PdfWriter from reportlab.lib.pagesizes import A4 def merge_4_invoices_to_a4(pdf_paths, output_path): writer = PdfWriter() a4_width, a4_height = A4 # 缩放比例:让单张发票占A4的1/4(2x2布局) scale = 0.5 invoice_width = a4_width * scale invoice_height = a4_height * scale # 按每4个PDF为一批处理 for batch_start in range(0, len(pdf_paths), 4): batch = pdf_paths[batch_start:batch_start+4] # 创建空白A4页面 a4_page = writer.add_blank_page(width=a4_width, height=a4_height) for idx, pdf_path in enumerate(batch): reader = PdfReader(pdf_path) invoice_page = reader.pages[0] # 计算当前发票在A4上的位置(2x2排列) row = idx // 2 col = idx % 2 # PDF坐标原点在左下角,所以y轴需要反向计算 x_pos = col * invoice_width y_pos = a4_height - (row + 1) * invoice_height # 缩放并将发票合并到A4页面的对应位置 a4_page.merge_scaled_translated_page(invoice_page, scale, x_pos, y_pos) # 保存合并后的PDF with open(output_path, "wb") as out_file: writer.write(out_file)
3. 修改main函数调用合并功能
在生成所有单个发票后,收集PDF路径并调用合并函数:
def main(): current_dir = os.path.abspath(os.getcwd()) output_subdir = "Output" merged_subdir = "Merged_A4" pdf_output_path = os.path.join(current_dir, output_subdir) merged_output_path = os.path.join(current_dir, merged_subdir) # 创建输出目录 if not os.path.exists(pdf_output_path): os.makedirs(pdf_output_path) if not os.path.exists(merged_output_path): os.makedirs(merged_output_path) excel_path = os.path.join(current_dir, "Invoice.xlsx") image_path = os.path.join(current_dir, "logo.jpg") qr_code_path = os.path.join(current_dir, "QR_code.jpg") # 生成所有单个发票PDF generate_bills(excel_path, pdf_output_path, image_path, qr_code_path) # 收集所有生成的发票PDF路径 pdf_paths = [] for filename in os.listdir(pdf_output_path): if filename.endswith(".pdf"): pdf_paths.append(os.path.join(pdf_output_path, filename)) # 合并为A4页面的PDF merge_4_invoices_to_a4(pdf_paths, os.path.join(merged_output_path, "merged_invoices.pdf")) if __name__ == '__main__': main()
补充说明
- 如果原有发票尺寸不适合2x2缩放,可调整
scale参数,或手动计算缩放比例确保发票完整显示; - 若想直接用reportlab绘制4张发票,需修改
create_pdf函数,添加x_offset和y_offset参数,所有绘制元素的坐标都加上偏移量,但这种方法需要改动大量硬编码坐标,不如合并PDF高效。
内容的提问来源于stack exchange,提问作者omkar
相关产品推荐
相关产品推荐

