Python解析AWS SES转发邮件Base64附件失败及嵌套转发处理问题
解决方案:修复附件提取及嵌套转发邮件处理问题
针对你遇到的Base64附件提取失败、嵌套转发邮件附件丢失问题,以下是具体修复方案:
核心问题分析
- 嵌套转发附件丢失:原代码处理
message/rfc822类型的转发邮件时,仅调用extract_email_info但未将返回的附件列表合并到主邮件数据中,导致转发邮件的附件被丢弃。 - 附件判断条件不全:原代码的附件识别逻辑未覆盖所有场景(比如内联附件、图片类附件等),部分Base64编码的附件因不符合判断条件未被捕获。
关键修复步骤
1. 合并转发邮件的附件及信息
修改process_part中对message/rfc822类型的处理逻辑,将转发邮件的附件合并到主邮件的附件列表,并可选存储转发邮件的完整信息用于审核:
if content_type == "message/rfc822": forwarded_msg = email.message_from_bytes(part.get_payload(decode=True)) forwarded_data = extract_email_info(forwarded_msg) # 合并转发邮件的附件到主附件列表 email_data['attachments'].extend(forwarded_data['attachments']) # 存储转发邮件的完整信息(可选,用于审核场景) if 'forwarded_emails' not in email_data: email_data['forwarded_emails'] = [] email_data['forwarded_emails'].append(forwarded_data)
2. 优化附件识别逻辑
扩展附件判断条件,覆盖更多附件类型(如图片、音频、内联附件等),确保Base64编码的附件能被正确识别:
elif (part.get_filename() is not None) or \ (("attachment" in disposition or "inline" in disposition) and part.get_filename()) or \ content_type.startswith(('application/', 'image/', 'audio/', 'video/')): attachment = { 'file_name': part.get_filename() or f"unnamed_{content_type.replace('/', '_')}", 'content_type': content_type, 'data': part.get_payload(decode=True), # 自动解码Base64等编码 'content_id': part.get('Content-ID'), 'disposition': disposition, } email_data['attachments'].append(attachment)
3. 补充extract_date函数实现
原代码中extract_date未定义,添加标准的邮件日期解析逻辑:
from email.utils import parsedate_to_datetime def extract_date(msg): date_str = msg.get('Date') if not date_str: return None try: # 解析为ISO格式字符串,方便存储和处理 return parsedate_to_datetime(date_str).isoformat() except (TypeError, ValueError): # 解析失败时返回原始日期字符串 return date_str
完整修复后的代码
import os import base64 import email from email.utils import parsedate_to_datetime def extract_date(msg): date_str = msg.get('Date') if not date_str: return None try: return parsedate_to_datetime(date_str).isoformat() except (TypeError, ValueError): return date_str def extract_email_info(msg): # Extract basic email details email_data = { 'sender': msg.get('From', 'Unknown Sender'), 'recipient': msg.get('To', 'Unknown Recipient'), 'subject': msg.get('Subject', 'No Subject'), 'date': extract_date(msg), 'body_plain': None, 'attachments': [], } for part in msg.walk(): process_part(part, email_data) # Ensure plain text body has a fallback message if not email_data['body_plain']: email_data['body_plain'] = "No plain text content found" return email_data def process_part(part, email_data): content_type = part.get_content_type() disposition = str(part.get("Content-Disposition", "")) if content_type == "message/rfc822": forwarded_msg = email.message_from_bytes(part.get_payload(decode=True)) forwarded_data = extract_email_info(forwarded_msg) # 合并转发邮件的附件 email_data['attachments'].extend(forwarded_data['attachments']) # 存储转发邮件信息(可选) if 'forwarded_emails' not in email_data: email_data['forwarded_emails'] = [] email_data['forwarded_emails'].append(forwarded_data) elif content_type == "text/plain" and "attachment" not in disposition: if not email_data['body_plain']: email_data['body_plain'] = part.get_payload(decode=True).decode(part.get_content_charset() or 'utf-8') elif (part.get_filename() is not None) or \ (("attachment" in disposition or "inline" in disposition) and part.get_filename()) or \ content_type.startswith(('application/', 'image/', 'audio/', 'video/')): attachment = { 'file_name': part.get_filename() or f"unnamed_{content_type.replace('/', '_')}", 'content_type': content_type, 'data': part.get_payload(decode=True), 'content_id': part.get('Content-ID'), 'disposition': disposition, } email_data['attachments'].append(attachment) # 使用示例 # msg = email.message_from_string(raw_string) # result = extract_email_info(msg)
额外调试建议
如果仍有附件未被提取,可以在process_part开头添加打印语句,排查每个邮件part的属性:
print(f"Part Info: content_type={content_type}, disposition={disposition}, filename={part.get_filename()}, encoding={part.get('Content-Transfer-Encoding')}")
通过输出可以确认附件是否符合判断条件,以及编码是否正确。
内容的提问来源于stack exchange,提问作者ilbets
相关产品推荐
相关产品推荐

