You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python调用Gmail API:如何将邮件正文解码为可读文本?

你的代码存在几个导致解码乱码的问题:

  • decode_base64_data仅完成Base64解码,返回的是字节流而非字符串,直接打印会显示b'xxx'格式的内容。
  • 未处理邮件正文的字符编码(比如UTF-8、GBK等),也没考虑quoted-printable这类常见的邮件编码格式。
  • 仅取第一个邮件Part,未处理嵌套的多部分邮件结构。

以下是修复后的完整代码:

from __future__ import print_function
import os.path
import re
import base64
import email
from email import policy
from email.parser import BytesParser
from google.auth.transport.requests import Request
from google.oauth2.credentials import Credentials
from google_auth_oauthlib.flow import InstalledAppFlow
from googleapiclient.discovery import build
from googleapiclient.errors import HttpError


SCOPES = ['https://www.googleapis.com/auth/gmail.readonly']


def main():
    creds = get_credentials()

    try:
        service = build('gmail', 'v1', credentials=creds)
        messages_list = service.users().messages().list(userId='me').execute().get('messages', [])
        for msg in messages_list:
            # 获取原始邮件数据,便于用email库解析
            txt = service.users().messages().get(userId='me', id=msg['id'], format='raw').execute()
            msg_bytes = base64.urlsafe_b64decode(txt['raw'])
            # 解析邮件结构
            parsed_msg = BytesParser(policy=policy.default).parsebytes(msg_bytes)
            
            sender = parsed_msg['From']
            if re.search("^Domo", sender):
                # 获取可读的邮件正文
                body = get_email_body(parsed_msg)
                print(body)

    except HttpError as error:
        handle_api_error(error)


def get_credentials():
    creds = None
    token_file = 'token.json'

    if os.path.exists(token_file):
        creds = Credentials.from_authorized_user_file(token_file, SCOPES)

    if not creds or not creds.valid:
        creds = refresh_credentials(creds, token_file)

    save_credentials(creds, token_file)
    return creds


def refresh_credentials(creds, token_file):
    if creds and creds.expired and creds.refresh_token:
        creds.refresh(Request())
    else:
        flow = InstalledAppFlow.from_client_secrets_file(
            'credentials.json', SCOPES)
        creds = flow.run_local_server(port=0)
    return creds


def save_credentials(creds, token_file):
    with open(token_file, 'w') as token:
        token.write(creds.to_json())


def get_email_body(parsed_msg):
    """解析邮件正文,优先返回纯文本,无纯文本则返回HTML"""
    if parsed_msg.is_multipart():
        # 遍历所有邮件部分,跳过附件
        for part in parsed_msg.walk():
            content_type = part.get_content_type()
            if content_type in ('text/plain', 'text/html') and not part.is_attachment():
                charset = part.get_content_charset() or 'utf-8'
                try:
                    return part.get_content().encode(charset).decode(charset)
                except Exception:
                    # 编码异常时用utf-8兜底,避免乱码
                    return part.get_content().encode('utf-8', errors='replace').decode('utf-8', errors='replace')
    else:
        # 非多部分邮件直接提取正文
        content_type = parsed_msg.get_content_type()
        charset = parsed_msg.get_content_charset() or 'utf-8'
        try:
            return parsed_msg.get_content().encode(charset).decode(charset)
        except Exception:
            return parsed_msg.get_content().encode('utf-8', errors='replace').decode('utf-8', errors='replace')


def handle_api_error(error):
    print(f'An error occurred: {error}')


if __name__ == '__main__':
    main()

关键修改说明

  1. 改用format='raw'获取原始邮件数据,借助Python内置的email库解析邮件结构,比手动处理Part更可靠。
  2. 新增get_email_body函数,递归遍历所有邮件Part,自动识别纯文本/HTML内容,同时处理字符编码异常。
  3. 利用part.get_content()方法自动处理Base64、quoted-printable等编码格式,无需手动解码。

内容的提问来源于stack exchange,提问作者Tito Lulu

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.20 18:24:59