You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

邮件正文提取后的间距异常问题及代码求助

邮件正文提取后丢失换行的修复方案

问题场景

提取邮件正文分享至其他通讯平台时,原内容的换行全部消失,示例如下:

原邮件末尾内容:
Regards
Abcde Xyz
Dir , PD
Software Corp | Mumbai
M : + 91 9876543210

提取后结果:
RegardsAbcde XyzDir , PD Software Corp | MumbaiM : + 91 9876543210

用户原始代码

function conVert() {
        Office.context.mailbox.item.body.getAsync(Office.CoercionType.Html, function (result) {
            if (result.status === Office.AsyncResultStatus.Succeeded) {
                let htmlBody = result.value;

                // Remove the comment block from the HTML body
                let modifiedHtmlBody = htmlBody.replace(/<!--[\s\S]*?-->/g, '');

                // Temporary div element to parse the modified HTML content
                let tempDiv = document.createElement('div');
                tempDiv.innerHTML = modifiedHtmlBody;

                // Convert anchor tags to plain text, preserving the URLs as clickable links
                let anchorTags = tempDiv.getElementsByTagName('a');
                for (let i = 0; i < anchorTags.length; i++) {
                    let anchorTag = anchorTags[i];
                    let url = anchorTag.getAttribute('href');
                    anchorTag.textContent = url;
                    anchorTag.setAttribute('href', url);
                }

                // Convert the modified HTML content to plain text
                let plainTextBody = tempDiv.textContent;

                // Paste the plain text body content into the textarea with id "item-Content"
                let textarea = document.getElementById("item-Content");
                textarea.value = plainTextBody;
            }
        });
    }

   conVert();

问题原因

tempDiv.textContent会直接提取HTML中的文本内容,但完全忽略所有HTML结构(包括<p>、<br>这类换行相关标签),导致所有文本挤在一起。

修复方案

在转换为纯文本前,先将HTML中的换行相关标签替换成对应的换行符,再提取文本。具体修改如下:

修改后的代码

function conVert() {
    Office.context.mailbox.item.body.getAsync(Office.CoercionType.Html, function (result) {
        if (result.status === Office.AsyncResultStatus.Succeeded) {
            let htmlBody = result.value;

            // 移除HTML注释
            let modifiedHtmlBody = htmlBody.replace(/<!--[\s\S]*?-->/g, '');
            
            // 处理换行:将<p>标签替换为两个换行(段落间隔),<br>替换为单个换行
            modifiedHtmlBody = modifiedHtmlBody
                .replace(/<p[^>]*>/g, '\n\n')
                .replace(/<br[^>]*>/g, '\n')
                .replace(/<\/p>/g, '');

            // 创建临时div解析处理后的HTML
            let tempDiv = document.createElement('div');
            tempDiv.innerHTML = modifiedHtmlBody;

            // 处理链接:将a标签替换为纯文本链接
            let anchorTags = tempDiv.getElementsByTagName('a');
            for (let i = anchorTags.length - 1; i >= 0; i--) { // 反向遍历避免索引错乱
                let anchorTag = anchorTags[i];
                let url = anchorTag.getAttribute('href');
                let textNode = document.createTextNode(url);
                anchorTag.parentNode.replaceChild(textNode, anchorTag);
            }

            // 提取纯文本并去除首尾多余空白
            let plainTextBody = tempDiv.textContent.trim();

            // 写入textarea
            let textarea = document.getElementById("item-Content");
            textarea.value = plainTextBody;
        }
    });
}

conVert();

关键修改点说明

  • 提前处理HTML中的<p>和<br>标签,将其转换为对应的换行符(<p>对应两个换行,模拟段落间的空行;<br>对应单个换行)。
  • 反向遍历处理<a>标签,避免删除元素后导致的索引错乱,直接将标签替换为纯文本链接(避免保留HTML结构影响文本提取)。
  • 最后用trim()去除文本首尾可能存在的多余空白。

内容的提问来源于stack exchange,提问作者Pearl

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 12:05:19