You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Apache POI提取Docx中嵌入Zip/Exe文件的问题咨询

问题描述

我使用XWPFDocument解析包含Zip、.exe、.docx、.xlsx嵌入文件的Word Docx文档,可通过XWPFDocument的PackagePart成功提取.docx和.xlsx文件,但Zip和.exe文件在提取时被转换为oleObject[1...4].bin,无法还原原文件名与属性。对比原.exe文件与代码生成的oleObject.bin的MD5值,二者完全不同。我无法从XWPFDocument获取嵌入.exe、Zip文件的原文件名属性,也无法提取到原始文件。现寻求可实现提取Docx中完整.exe或Zip文件的代码示例或代码片段。

当前使用的代码:

public void extractEmbeddedFiles(String filePath) throws IOException, InvalidFormatException {

    InputStream is = getFileFromResourceAsStream(filePath);

    OPCPackage opcPackage = OPCPackage.open(is);

    List<PackagePart> packageParts = opcPackage.getParts();

    for (PackagePart packagePart : packageParts) {
        System.out.println(packagePart.getPartName().toString());

        if (packagePart.getPartName().getName().contains("/word/")) {
            try {
                InputStream is1 = packagePart.getInputStream();
                long n = 254;
                is1.skip(n);
                WordExtractorTest test = new WordExtractorTest();
                test.test(packagePart);

            } catch (Exception e) {
                System.out.println("Exception occurred :" + e.getMessage());
            }
        }
    }

}

解决方案

问题本质

.exe、.zip这类非Office格式文件是以OLE对象的形式嵌入Word文档的,它们被包裹在OLE容器结构中。直接读取PackagePart的流会包含OLE头信息,导致提取出的文件与原文件MD5不一致;而原文件名等元数据存储在OLE对象的属性流里,必须解析OLE结构才能获取。

完整提取代码示例

以下代码利用POI的OLE解析工具,实现完整提取所有嵌入文件(包括.exe、.zip)并还原原文件名:

import org.apache.poi.poifs.filesystem.DirectoryEntry;
import org.apache.poi.poifs.filesystem.DocumentEntry;
import org.apache.poi.poifs.filesystem.POIFSFileSystem;
import org.apache.poi.xwpf.usermodel.XWPFDocument;
import org.apache.poi.xwpf.usermodel.XWPFPictureData;
import org.apache.poi.xwpf.usermodel.XWPFObject;
import org.apache.poi.xwpf.usermodel.XWPFParagraph;

import java.io.*;
import java.util.List;

public class EmbeddedFileExtractor {

    public void extractAllEmbeddedFiles(String docxPath, String outputDir) throws IOException {
        File outputDirectory = new File(outputDir);
        if (!outputDirectory.exists()) {
            outputDirectory.mkdirs();
        }

        try (FileInputStream fis = new FileInputStream(docxPath);
             XWPFDocument doc = new XWPFDocument(fis)) {

            // 提取段落中的OLE对象(exe/zip等非Office文件)
            for (XWPFParagraph paragraph : doc.getParagraphs()) {
                List<XWPFObject> oleObjects = paragraph.getEmbeddedObjects();
                for (XWPFObject oleObject : oleObjects) {
                    extractOLEObject(oleObject, outputDirectory);
                }
            }

            // 提取原生Office嵌入文件(docx/xlsx/pptx等)
            List<XWPFPictureData> embeddedOfficeFiles = doc.getAllPackagePictures();
            for (XWPFPictureData data : embeddedOfficeFiles) {
                String fileName = data.getFileName();
                if (fileName == null || fileName.isEmpty()) {
                    // 从ContentType推导默认文件名
                    String contentType = data.getPackagePart().getContentType();
                    String ext = switch (contentType) {
                        case "application/vnd.openxmlformats-officedocument.wordprocessingml.document" -> ".docx";
                        case "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" -> ".xlsx";
                        case "application/vnd.openxmlformats-officedocument.presentationml.presentation" -> ".pptx";
                        default -> ".bin";
                    };
                    fileName = "embedded_office_" + System.currentTimeMillis() + ext;
                }
                try (OutputStream os = new FileOutputStream(new File(outputDirectory, fileName))) {
                    os.write(data.getData());
                }
            }
        }
    }

    private void extractOLEObject(XWPFObject oleObject, File outputDir) throws IOException {
        try (InputStream is = oleObject.getPackagePart().getInputStream();
             POIFSFileSystem poifs = new POIFSFileSystem(is)) {

            DirectoryEntry root = poifs.getRoot();
            // 遍历OLE容器内的条目,提取原始文件
            for (org.apache.poi.poifs.filesystem.Entry entry : root) {
                if (entry instanceof DocumentEntry docEntry) {
                    String fileName = getOriginalFileNameFromOLE(poifs);
                    if (fileName == null || fileName.isEmpty()) {
                        fileName = "extracted_ole_" + System.currentTimeMillis() + getFileExtensionFromOLE(docEntry);
                    }

                    // 写入原始文件内容
                    try (InputStream docIs = docEntry.getInputStream();
                         OutputStream os = new FileOutputStream(new File(outputDir, fileName))) {
                        byte[] buffer = new byte[4096];
                        int bytesRead;
                        while ((bytesRead = docIs.read(buffer)) != -1) {
                            os.write(buffer, 0, bytesRead);
                        }
                    }
                }
            }
        }
    }

    private String getOriginalFileNameFromOLE(POIFSFileSystem poifs) {
        try {
            // 尝试从Ole10Native结构读取原文件名(最可靠的方式)
            DirectoryEntry root = poifs.getRoot();
            if (root.hasEntry("Ole10Native")) {
                DocumentEntry oleNativeEntry = (DocumentEntry) root.getEntry("Ole10Native");
                try (InputStream is = oleNativeEntry.getInputStream()) {
                    byte[] buffer = new byte[(int) oleNativeEntry.getSize()];
                    is.read(buffer);
                    // Ole10Native结构:前20字节为头部,之后是文件名长度+文件名
                    int nameOffset = 20;
                    int nameLength = buffer[nameOffset] & 0xFF;
                    if (nameLength > 0 && nameOffset + nameLength < buffer.length) {
                        return new String(buffer, nameOffset + 1, nameLength - 1, "GBK");
                    }
                }
            }
            //  fallback:从SummaryInformation读取标题作为文件名
            if (root.hasEntry("\u0005SummaryInformation")) {
                DocumentEntry infoEntry = (DocumentEntry) root.getEntry("\u0005SummaryInformation");
                try (InputStream is = infoEntry.getInputStream()) {
                    org.apache.poi.hpsf.SummaryInformation si = new org.apache.poi.hpsf.SummaryInformation(is);
                    return si.getTitle();
                }
            }
        } catch (Exception e) {
            e.printStackTrace();
        }
        return null;
    }

    private String getFileExtensionFromOLE(DocumentEntry docEntry) {
        String entryName = docEntry.getName().toUpperCase();
        if (entryName.contains("EXE")) {
            return ".exe";
        } else if (entryName.contains("ZIP")) {
            return ".zip";
        }
        return ".bin";
    }
}

核心说明

  1. OLE容器解析:通过POIFSFileSystem解析OLE流,剥离OLE包装层,直接读取内部的原始文件内容,保证MD5与原文件一致。
  2. 文件名还原:优先从OLE的Ole10Native结构读取原文件名,这是嵌入非Office文件时存储文件名的标准位置;如果失败则尝试从SummaryInformation获取。
  3. 分类型处理:区分OLE类型嵌入文件和原生Office嵌入文件,确保所有类型的嵌入文件都能正确提取。

使用示例

public static void main(String[] args) {
    try {
        EmbeddedFileExtractor extractor = new EmbeddedFileExtractor();
        extractor.extractAllEmbeddedFiles("你的文档路径.docx", "输出目录路径");
        System.out.println("所有嵌入文件提取完成");
    } catch (IOException e) {
        e.printStackTrace();
    }
}

内容的提问来源于stack exchange,提问作者Code Trickle

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.16 16:57:45