You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

控制器接收含嵌套ZIP的文件,如何提取所有XML生成扁平目录?

递归提取嵌套ZIP中的XML并生成扁平目录

核心逻辑

不管ZIP嵌套多少层,只聚焦两件事:

  • 遇到ZIP文件就拆包,递归处理内部内容
  • 遇到XML文件就提取到统一目录,同名文件自动加序号避免覆盖

Java实现(Spring Boot控制器场景)

先写负责递归提取的工具类:

import java.io.*;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.Enumeration;
import java.util.zip.ZipEntry;
import java.util.zip.ZipFile;

public class XmlExtractor {
    // 自定义XML输出目录
    private static final String TARGET_DIR = "./extracted-xmls/";

    public static void extractXmlsFromZip(File zipFile) throws IOException {
        // 确保输出目录存在
        Files.createDirectories(Paths.get(TARGET_DIR));

        try (ZipFile zf = new ZipFile(zipFile)) {
            Enumeration<? extends ZipEntry> entries = zf.entries();
            while (entries.hasMoreElements()) {
                ZipEntry entry = entries.nextElement();
                if (entry.isDirectory()) continue;

                String entryName = entry.getName();
                // 碰到嵌套ZIP就递归处理
                if (entryName.endsWith(".zip")) {
                    File tempZip = File.createTempFile("temp_", ".zip");
                    try (InputStream is = zf.getInputStream(entry);
                         OutputStream os = new FileOutputStream(tempZip)) {
                        byte[] buffer = new byte[1024];
                        int len;
                        while ((len = is.read(buffer)) != -1) os.write(buffer, 0, len);
                    }
                    extractXmlsFromZip(tempZip);
                    tempZip.delete(); // 用完删除临时文件
                }
                // 碰到XML就提取到目标目录
                else if (entryName.endsWith(".xml")) {
                    String fileName = Paths.get(entryName).getFileName().toString();
                    Path targetPath = Paths.get(TARGET_DIR, fileName);
                    // 同名文件自动加序号
                    int counter = 1;
                    while (Files.exists(targetPath)) {
                        String baseName = fileName.substring(0, fileName.lastIndexOf("."));
                        String extension = fileName.substring(fileName.lastIndexOf("."));
                        fileName = baseName + "_" + counter + extension;
                        targetPath = Paths.get(TARGET_DIR, fileName);
                        counter++;
                    }
                    // 写入XML文件
                    try (InputStream is = zf.getInputStream(entry);
                         OutputStream os = Files.newOutputStream(targetPath)) {
                        byte[] buffer = new byte[1024];
                        int len;
                        while ((len = is.read(buffer)) != -1) os.write(buffer, 0, len);
                    }
                }
            }
        }
    }
}

再在控制器中调用工具类:

import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RestController;
import org.springframework.web.multipart.MultipartFile;
import java.io.File;
import java.io.IOException;

@RestController
public class ZipProcessingController {

    @PostMapping("/process-zip")
    public String processZip(@RequestParam("file") MultipartFile zipFile) {
        if (zipFile.isEmpty()) return "请上传有效的ZIP文件";

        try {
            // 将上传的MultipartFile转为临时文件
            File tempFile = File.createTempFile("upload_", ".zip");
            zipFile.transferTo(tempFile);

            // 执行提取逻辑
            XmlExtractor.extractXmlsFromZip(tempFile);

            tempFile.delete(); // 清理临时文件
            return "XML提取完成,所有文件已保存至:" + XmlExtractor.TARGET_DIR;
        } catch (IOException e) {
            e.printStackTrace();
            return "处理失败:" + e.getMessage();
        }
    }
}

Python实现(Flask控制器场景)

先写提取逻辑函数:

import zipfile
import os
import tempfile
from pathlib import Path

# 自定义XML输出目录
TARGET_DIR = "./extracted-xmls/"

def extract_xmls_from_zip(zip_path):
    os.makedirs(TARGET_DIR, exist_ok=True)

    with zipfile.ZipFile(zip_path, 'r') as zf:
        for entry in zf.infolist():
            if entry.is_dir(): continue

            entry_name = entry.filename
            # 处理嵌套ZIP
            if entry_name.endswith('.zip'):
                with tempfile.NamedTemporaryFile(suffix='.zip', delete=False) as temp_zip:
                    temp_zip.write(zf.read(entry))
                    temp_zip_path = temp_zip.name
                extract_xmls_from_zip(temp_zip_path)
                os.unlink(temp_zip_path)
            # 提取XML文件
            elif entry_name.endswith('.xml'):
                file_name = Path(entry_name).name
                target_path = os.path.join(TARGET_DIR, file_name)
                # 同名文件自动加序号
                counter = 1
                while os.path.exists(target_path):
                    base, ext = os.path.splitext(file_name)
                    file_name = f"{base}_{counter}{ext}"
                    target_path = os.path.join(TARGET_DIR, file_name)
                    counter += 1
                # 写入文件
                with open(target_path, 'wb') as out_file:
                    out_file.write(zf.read(entry))

再写Flask控制器:

from flask import Flask, request, jsonify

app = Flask(__name__)

@app.route('/process-zip', methods=['POST'])
def process_zip():
    if 'file' not in request.files:
        return jsonify({"error": "请上传ZIP文件"}), 400
    zip_file = request.files['file']
    if zip_file.filename == '':
        return jsonify({"error": "请选择有效的ZIP文件"}), 400
    if not zip_file.filename.endswith('.zip'):
        return jsonify({"error": "仅支持ZIP文件"}), 400

    with tempfile.NamedTemporaryFile(suffix='.zip', delete=False) as temp_file:
        zip_file.save(temp_file)
        temp_file_path = temp_file.name

    try:
        extract_xmls_from_zip(temp_file_path)
        return jsonify({"message": f"XML提取完成,文件已保存至{TARGET_DIR}"}), 200
    except Exception as e:
        return jsonify({"error": f"处理失败:{str(e)}"}), 500
    finally:
        os.unlink(temp_file_path) # 清理临时文件

if __name__ == '__main__':
    app.run(debug=True)

注意事项

  • 临时文件用完必须删除,避免磁盘占用过高
  • 必须处理文件名冲突,否则后续XML会覆盖之前的同名文件
  • 大文件场景优先用流处理(Java示例已实现),避免一次性读入内存导致OOM
  • 生产环境需补充更完善的异常捕获,比如ZIP损坏、权限不足等情况

内容的提问来源于stack exchange,提问作者VaheCh

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.31 17:45:28