使用Apache POI合并Docx:不同字体文档格式错乱问题排查
问题分析与解决
你的代码核心问题是只拼接了文档正文的XML内容,完全没处理样式表的合并。Docx格式里,字体、段落样式这些格式信息不是存在正文里的,而是单独存储在文档的样式部分(styles.xml)。当第二个文档使用了第一个文档没有的字体或样式时,合并后的文档根本找不到这些样式的定义,Word只能自动用默认样式替代,自然就出现格式错乱的情况。
要正确合并带不同字体的Docx,必须同时完成两件事:
- 把第二个文档的样式导入到第一个文档,确保所有需要的格式定义都存在
- 逐个复制第二个文档的正文元素(段落、表格等)到第一个文档,同时修正样式引用的对应关系
修改后的完整代码
import org.apache.poi.xwpf.usermodel.*; import org.openxmlformats.schemas.wordprocessingml.x2006.main.*; import java.io.FileInputStream; import java.io.FileOutputStream; import java.io.IOException; import java.util.HashMap; import java.util.Map; public class DocxMerger { public static void main(String[] args) { try (FileInputStream fis1 = new FileInputStream("path/to/your/first_document.docx"); XWPFDocument doc1 = new XWPFDocument(fis1); FileInputStream fis2 = new FileInputStream("path/to/your/second_document.docx"); XWPFDocument doc2 = new XWPFDocument(fis2)) { // 先导入第二个文档的样式到第一个文档 Map<String, String> styleIdMap = importStyles(doc1, doc2); // 然后逐个复制第二个文档的正文元素 copyBodyElements(doc1, doc2, styleIdMap); // 保存合并后的文档 try (FileOutputStream fos = new FileOutputStream("path/to/your/combined_document.docx")) { doc1.write(fos); } } catch (IOException e) { e.printStackTrace(); } } // 导入第二个文档的样式到第一个文档,返回样式ID的映射关系(原ID -> 新ID) private static Map<String, String> importStyles(XWPFDocument targetDoc, XWPFDocument sourceDoc) { Map<String, String> styleIdMap = new HashMap<>(); XWPFStyles targetStyles = targetDoc.createStyles(); XWPFStyles sourceStyles = sourceDoc.getStyles(); for (XWPFStyle sourceStyle : sourceStyles.getStyles()) { // 跳过默认样式(避免重复) if (sourceStyle.getType() == STStyleType.DEFAULT) continue; String sourceStyleId = sourceStyle.getStyleId(); // 如果目标文档没有这个样式,就导入 if (targetStyles.getStyle(sourceStyleId) == null) { XWPFStyle newStyle = targetStyles.createStyle(); newStyle.setStyle(sourceStyle.getCTStyle()); styleIdMap.put(sourceStyleId, newStyle.getStyleId()); } else { // 如果样式已存在,直接映射到现有ID styleIdMap.put(sourceStyleId, sourceStyleId); } } return styleIdMap; } // 复制第二个文档的所有正文元素到第一个文档,同时修正样式引用 private static void copyBodyElements(XWPFDocument targetDoc, XWPFDocument sourceDoc, Map<String, String> styleIdMap) { // 复制段落 for (XWPFParagraph sourcePara : sourceDoc.getParagraphs()) { XWPFParagraph newPara = targetDoc.createParagraph(); copyParagraph(sourcePara, newPara, styleIdMap); } // 复制表格 for (XWPFTable sourceTable : sourceDoc.getTables()) { XWPFTable newTable = targetDoc.createTable(); copyTable(sourceTable, newTable, styleIdMap); } } // 复制单个段落,处理样式映射 private static void copyParagraph(XWPFParagraph sourcePara, XWPFParagraph targetPara, Map<String, String> styleIdMap) { // 映射段落样式ID String sourceStyleId = sourcePara.getStyleID(); if (sourceStyleId != null && styleIdMap.containsKey(sourceStyleId)) { targetPara.setStyleID(styleIdMap.get(sourceStyleId)); } // 复制段落属性 targetPara.getCTP().set(sourcePara.getCTP().copy()); // 复制每个run的内容和样式 for (XWPFRun sourceRun : sourcePara.getRuns()) { XWPFRun newRun = targetPara.createRun(); copyRun(sourceRun, newRun); } } // 复制单个表格,处理单元格内的段落样式 private static void copyTable(XWPFTable sourceTable, XWPFTable targetTable, Map<String, String> styleIdMap) { targetTable.getCTTbl().set(sourceTable.getCTTbl().copy()); // 遍历表格的行和单元格,修正段落样式 for (int i = 0; i < sourceTable.getRows().size(); i++) { XWPFTableRow sourceRow = sourceTable.getRow(i); XWPFTableRow targetRow = targetTable.getRow(i); for (int j = 0; j < sourceRow.getTableCells().size(); j++) { XWPFTableCell sourceCell = sourceRow.getCell(j); XWPFTableCell targetCell = targetRow.getCell(j); // 清空目标单元格原有内容 targetCell.removeParagraph(0); for (XWPFParagraph sourcePara : sourceCell.getParagraphs()) { XWPFParagraph newPara = targetCell.addParagraph(); copyParagraph(sourcePara, newPara, styleIdMap); } } } } // 复制单个文本run的内容和格式 private static void copyRun(XWPFRun sourceRun, XWPFRun targetRun) { // 复制文本内容 targetRun.setText(sourceRun.getText(0)); // 复制字体格式 targetRun.setFontFamily(sourceRun.getFontFamily()); targetRun.setFontSize(sourceRun.getFontSize()); targetRun.setBold(sourceRun.isBold()); targetRun.setItalic(sourceRun.isItalic()); targetRun.setUnderline(sourceRun.getUnderline()); targetRun.setColor(sourceRun.getColor()); // 复制其他run属性 CTR ctr = sourceRun.getCTR(); if (ctr.isSetRPr()) { targetRun.getCTR().setRPr(ctr.getRPr().copy()); } } }
代码说明
- 先处理样式表:把第二个文档中第一个文档没有的样式全部导入,避免样式缺失导致格式失效
- 逐个复制正文元素:段落、表格都会单独复制,同时修正样式ID的引用关系,确保格式能正确对应
- 细节处理:文本run的字体、粗体、斜体、下划线等格式都会完整复制,保留原始文档的排版细节
内容的提问来源于stack exchange,提问作者AI_techno
相关产品推荐
相关产品推荐

