You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过MS GraphAPI获取带附件的最新邮件会话?

解决MS Graph API获取邮件会话最新回复及附件的问题

实现方案

1. 利用会话ID定位最新回复

每封邮件的conversationId字段对应所属会话,通过该ID可以查询会话内所有邮件,按receivedDateTime倒序排序后,第一条就是最新的回复邮件。

2. 提取纯最新内容(去除历史引用)

Outlook的回复邮件中,历史内容通常会被包裹在<blockquote>标签或带有特定类名的容器中(比如div.OriginalMessage),可以通过BeautifulSoup过滤掉这些部分,只保留用户新增的内容。

3. 获取邮件附件

通过Graph API的/messages/{messageId}/attachments端点,可获取对应邮件的所有附件,支持区分普通附件和内嵌附件。

修改后的完整代码

from bs4 import BeautifulSoup
import requests

# 替换为你的配置参数
MICROSOFT_TENANT_ID = "你的租户ID"
MICROSOFT_CLIENT_ID = "你的客户端ID"
MICROSOFT_CLIENT_SECRET = "你的客户端密钥"
MICROSOFT_SCOPES = "https://graph.microsoft.com/.default"
MICROSOFT_USERNAME = "你的邮箱账号"
MICROSOFT_PASSWORD = "你的邮箱密码"

def get_access_token():
    try:
        token_url = f"https://login.microsoftonline.com/{MICROSOFT_TENANT_ID}/oauth2/v2.0/token"
        data = {
            'client_id': MICROSOFT_CLIENT_ID,
            'client_secret': MICROSOFT_CLIENT_SECRET,
            'grant_type': 'password',
            'scope': MICROSOFT_SCOPES,
            'username': MICROSOFT_USERNAME,
            'password': MICROSOFT_PASSWORD,
        }
        response = requests.post(token_url, data=data)
        if response.status_code == 200:
            return response.json().get('access_token')
        else:
            print(f"获取令牌失败: {response.text}")
            return None
    except Exception as e:
        print(f"获取令牌出错: {e}")
        return None

def fetch_latest_conversation_message(conversation_id, access_token):
    """获取会话中的最新邮件"""
    try:
        headers = {'Authorization': f'Bearer {access_token}'}
        # 查询会话内所有邮件,按接收时间倒序取第一条
        url = f'https://graph.microsoft.com/v1.0/me/messages?$filter=conversationId eq \'{conversation_id}\'&$orderby=receivedDateTime desc&$top=1'
        response = requests.get(url, headers=headers)
        if response.status_code == 200:
            return response.json().get('value')[0]
        else:
            print(f"获取会话邮件失败: {response.text}")
            return None
    except Exception as e:
        print(f"获取会话邮件出错: {e}")
        return None

def extract_latest_content(html_content):
    """提取邮件中的纯最新内容,去除历史引用"""
    try:
        soup = BeautifulSoup(html_content, 'html.parser')
        # 移除Outlook的历史引用块(blockquote或特定类容器)
        for blockquote in soup.find_all('blockquote'):
            blockquote.decompose()
        for original_msg in soup.find_all('div', class_='OriginalMessage'):
            original_msg.decompose()
        # 清理多余的空标签和换行
        clean_content = soup.get_text(strip=True, separator='\n')
        return clean_content
    except Exception as e:
        print(f"提取内容出错: {e}")
        return html_content

def fetch_email_attachments(message_id, access_token):
    """获取邮件的所有附件"""
    try:
        headers = {'Authorization': f'Bearer {access_token}'}
        url = f'https://graph.microsoft.com/v1.0/me/messages/{message_id}/attachments'
        response = requests.get(url, headers=headers)
        if response.status_code == 200:
            attachments = response.json().get('value', [])
            # 过滤掉内嵌附件(比如图片签名)
            return [att for att in attachments if not att.get('isInline', False)]
        else:
            print(f"获取附件失败: {response.text}")
            return []
    except Exception as e:
        print(f"获取附件出错: {e}")
        return []

def print_latest_conversation_details(folder):
    try:
        access_token = get_access_token()
        if not access_token:
            return
        
        # 先获取文件夹内最新的邮件(作为会话入口)
        headers = {'Authorization': f'Bearer {access_token}'}
        folder_url = f'https://graph.microsoft.com/v1.0/me/mailFolders/{folder}/messages?$top=1&$orderby=receivedDateTime desc'
        folder_response = requests.get(folder_url, headers=headers)
        if folder_response.status_code != 200:
            print(f"获取文件夹最新邮件失败: {folder_response.text}")
            return
        
        latest_email = folder_response.json().get('value')[0]
        conversation_id = latest_email.get('conversationId')
        if not conversation_id:
            print("该邮件不属于任何会话")
            return
        
        # 获取会话中的最新回复邮件
        latest_convo_msg = fetch_latest_conversation_message(conversation_id, access_token)
        if not latest_convo_msg:
            return
        
        # 打印邮件基本信息
        print(f"会话最新邮件主题: {latest_convo_msg.get('subject')}")
        print(f"发件人: {latest_convo_msg.get('from').get('emailAddress').get('address')}")
        print(f"接收时间: {latest_convo_msg.get('receivedDateTime')}")
        
        # 提取并打印纯最新内容
        body_html = latest_convo_msg.get('body', {}).get('content', '无内容')
        clean_content = extract_latest_content(body_html)
        print("\n最新回复内容:")
        print(clean_content)
        
        # 获取并打印附件信息
        attachments = fetch_email_attachments(latest_convo_msg.get('id'), access_token)
        if attachments:
            print("\n附件列表:")
            for idx, att in enumerate(attachments):
                print(f"{idx+1}. {att.get('name')} (大小: {att.get('size')}字节)")
                # 如需下载附件,可调用att.get('contentBytes')解码保存
        
    except Exception as e:
        print(f"处理会话邮件出错: {e}")

# 执行示例:获取收件箱和已发送邮件的会话最新内容
print("=== 收件箱会话最新内容 ===")
print_latest_conversation_details('inbox')
print("\n=== 已发送邮件会话最新内容 ===")
print_latest_conversation_details('sentitems')

关键说明

  • 会话定位:通过conversationId过滤邮件,确保获取的是同一会话内的所有邮件,再按时间排序取最新。
  • 内容提取:通过移除blockquote和OriginalMessage容器,直接得到用户新增的回复内容,比依赖<hr>标签更可靠(不同邮件客户端的分隔符可能不同)。
  • 附件处理:过滤掉isInline=True的内嵌附件,只保留用户主动添加的附件。

内容的提问来源于stack exchange,提问作者yabesh sam

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 12:02:20