如何用Python从Word文件提取图文并通过SMTP发送邮件
解决Python从Word提取图片并SMTP发送邮件的方案
一、从Word文档提取图片
使用python-docx库提取文档中的内嵌/浮动图片,先安装依赖:
pip install python-docx
提取图片的代码示例:
from docx import Document def extract_images_from_docx(doc_path): doc = Document(doc_path) image_list = [] # 提取内嵌图片 for inline_shape in doc.inline_shapes: if inline_shape.type == 3: # 类型3对应图片 image_bytes = inline_shape.image.blob image_list.append((f"inline_img_{len(image_list)}.png", image_bytes)) # 提取浮动图片 for shape in doc.shapes: if shape.shape_type == 13: # 类型13对应图片形状 image_bytes = shape.image.blob image_list.append((f"float_img_{len(image_list)}.png", image_bytes)) return image_list
二、构造带图片的邮件(避免编码过长问题)
不要手动拼接as_string,用MIMEMultipart管理邮件结构,支持两种图片发送方式:
方式1:图片嵌入邮件正文
通过CID关联图片,实现正文内显示:
import smtplib from email.mime.multipart import MIMEMultipart from email.mime.text import MIMEText from email.mime.image import MIMEImage from email.header import Header def send_embedded_image_email(smtp_server, smtp_port, sender, pwd, receiver, doc_text, images): msg = MIMEMultipart('related') msg['From'] = Header(sender, 'utf-8') msg['To'] = Header(receiver, 'utf-8') msg['Subject'] = Header('含Word图片的邮件', 'utf-8') # 构造HTML正文,通过CID引用图片 html_content = f"<p>{doc_text}</p>" for idx, (img_name, img_bytes) in enumerate(images): cid = f"img_{idx}" html_content += f'<p><img src="cid:{cid}" alt="{img_name}"></p>' # 添加图片到邮件 img_part = MIMEImage(img_bytes) img_part.add_header('Content-ID', f'<{cid}>') msg.attach(img_part) msg.attach(MIMEText(html_content, 'html', 'utf-8')) # 发送邮件 with smtplib.SMTP_SSL(smtp_server, smtp_port) as server: server.login(sender, pwd) server.sendmail(sender, receiver, msg.as_string())
方式2:图片作为附件发送
如果不需要内嵌,直接添加附件更简单:
def send_attachment_email(smtp_server, smtp_port, sender, pwd, receiver, doc_text, images): msg = MIMEMultipart() msg['From'] = Header(sender, 'utf-8') msg['To'] = Header(receiver, 'utf-8') msg['Subject'] = Header('Word图片附件邮件', 'utf-8') # 添加纯文本正文 msg.attach(MIMEText(doc_text, 'plain', 'utf-8')) # 添加图片附件 for img_name, img_bytes in images: img_part = MIMEImage(img_bytes) img_part.add_header('Content-Disposition', 'attachment', filename=img_name) msg.attach(img_part) with smtplib.SMTP_SSL(smtp_server, smtp_port) as server: server.login(sender, pwd) server.sendmail(sender, receiver, msg.as_string())
三、循环发送3-4封邮件
把发送逻辑封装后,循环调用即可:
def main(): # 配置信息 smtp_server = 'smtp.xxx.com' # 示例:smtp.qq.com、smtp.gmail.com smtp_port = 465 sender_email = 'your_email@xxx.com' sender_pwd = 'your_auth_code' # 用邮箱授权码,不是登录密码 receivers = ['recipient1@xxx.com', 'recipient2@xxx.com'] # 读取Word文本(你已实现的部分) doc_path = 'your_doc.docx' doc = Document(doc_path) doc_text = '\n'.join([para.text for para in doc.paragraphs]) # 提取图片 images = extract_images_from_docx(doc_path) # 循环发送4次 for i in range(4): receiver = receivers[i % len(receivers)] # 调用发送函数(可替换为send_attachment_email) send_embedded_image_email(smtp_server, smtp_port, sender_email, sender_pwd, receiver, f'{doc_text}\n第{i+1}封邮件', images) print(f'第{i+1}封邮件发送完成') if __name__ == '__main__': main()
关键提示
- 依赖
MIMEMultipart自动处理编码,彻底避免手动拼接as_string的长度/编码问题 - SMTP端口优先用465(SSL加密),邮箱密码需用授权码(在邮箱设置中开启POP3/SMTP后生成)
- 仅支持docx格式,doc文件需先转成docx再处理
内容的提问来源于stack exchange,提问作者Hasan Abdul Ghaffar
相关产品推荐
相关产品推荐

