如何在网页端使用JavaScript读取docx文件并保留文本换行符
前端读取docx保留换行的实现方案
方案1:基于现有docxtemplater依赖改造
docxtemplater内置的getFullText()方法默认会直接拼接所有段落文本,不会自动添加换行符,我们可以直接解析docx内的document.xml文件手动提取段落并追加换行,不需要引入额外依赖,修改后代码如下:
function openDocx(event) { var input = event.target; var reader = new FileReader(); reader.onload = async function () { var zip = new JSZip(reader.result); // 读取docx内的 document.xml 源文件 const xmlContent = await zip.file("word/document.xml").async("string"); const paragraphRegex = /<w:p[^>]*>(.*?)<\/w:p>/g; const textRegex = /<w:t[^>]*>(.*?)<\/w:t>/g; const paragraphList = []; let paragraphMatch; // 遍历所有段落节点 while ((paragraphMatch = paragraphRegex.exec(xmlContent)) !== null) { let paragraphText = ''; let textMatch; // 提取单个段落内的所有文本内容 while ((textMatch = textRegex.exec(paragraphMatch[1])) !== null) { paragraphText += textMatch[1]; } // 可根据需求选择是否过滤空段落 if (paragraphText) { paragraphList.push(paragraphText); } } // 用换行符拼接所有段落 const finalText = paragraphList.join('\n'); console.log(finalText); }; reader.readAsBinaryString(input.files[0]); };
方案2:改用专门的docx文本提取库mammoth.js
如果需要更稳定的兼容能力(比如处理软换行、特殊格式文本),可以直接放弃docxtemplater,改用mammoth.js实现,它内置的extractRawText方法默认就会保留段落换行,实现逻辑更简单:
function openDocx(event) { var input = event.target; var reader = new FileReader(); reader.onload = function () { mammoth.extractRawText({arrayBuffer: reader.result}) .then(res => { const finalText = res.value; console.log(finalText); }) .catch(err => { console.error('docx读取失败', err); }); }; reader.readAsArrayBuffer(input.files[0]); };
内容的提问来源于stack exchange,提问作者Robert
相关产品推荐
相关产品推荐

