You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何实现AWS S3大文件断网自动续传下载(Node/Python实现)

问题根源

原生pipe整段传输的方案无法捕获断网错误,核心原因有两点:

  • AWS SDK默认的流传输没有绑定底层socket的空闲超时检测,断网后TCP连接如果没有主动收到断开报文,会一直处于挂起等待状态,不会触发error事件
  • 整段传输没有用到S3的Range请求能力,即使捕获到错误也没法从断点位置续传,只能从头重新下载
Node.js 生产可用实现方案

这个方案实现了全链路错误捕获、断网自动重试续传、文件完整性校验,依赖AWS SDK v3(流错误处理比v2更稳定),先安装依赖:

npm install @aws-sdk/client-s3

完整代码如下:

const { S3Client, HeadObjectCommand, GetObjectCommand } = require("@aws-sdk/client-s3");
const fs = require("fs/promises");
const fss = require("fs");
const path = require("path");

// 配置项
const CONFIG = {
  accessKeyId: "",
  secretAccessKey: "",
  region: "",
  bucket: "",
  key: "",
  localPath: path.join("./", "file.zip"),
  timeout: 30000, // 30秒无数据判定为网络中断
  maxRetry: 10 // 最大重试次数
};

const s3Client = new S3Client({
  region: CONFIG.region,
  credentials: {
    accessKeyId: CONFIG.accessKeyId,
    secretAccessKey: CONFIG.secretAccessKey,
  },
  requestHandler: {
    connectionTimeout: 5000,
    socketTimeout: CONFIG.timeout
  }
});

// 触发重试的网络类错误码
const NETWORK_ERRORS = new Set(['ETIMEDOUT', 'ECONNRESET', 'ENOTFOUND', 'EPIPE', 'ECONNREFUSED', 'EAI_AGAIN']);

async function getLocalFileSize(filePath) {
  try {
    const stat = await fs.stat(filePath);
    return stat.size;
  } catch (e) {
    if (e.code === 'ENOENT') return 0;
    throw e;
  }
}

async function download() {
  // 获取S3文件总大小
  const headRes = await s3Client.send(new HeadObjectCommand({
    Bucket: CONFIG.bucket,
    Key: CONFIG.key
  }));
  const totalSize = headRes.ContentLength;
  console.log(`文件总大小: ${totalSize} bytes`);

  let retryCount = 0;
  let downloaded = await getLocalFileSize(CONFIG.localPath);

  // 本地文件已完整直接返回
  if (downloaded >= totalSize) {
    console.log("文件已存在且完整,无需下载");
    return;
  }

  while (downloaded < totalSize) {
    let timeoutTimer;
    try {
      console.log(`从 ${downloaded} 字节位置开始续传`);
      const getRes = await s3Client.send(new GetObjectCommand({
        Bucket: CONFIG.bucket,
        Key: CONFIG.key,
        Range: `bytes=${downloaded}-`
      }));

      const rs = getRes.Body;
      const ws = fss.createWriteStream(CONFIG.localPath, {
        flags: 'a', // 追加写入
        start: downloaded
      });

      // 统一错误处理
      const errorHandler = (err) => {
        rs.destroy();
        ws.destroy();
        throw err;
      };
      rs.on('error', errorHandler);
      ws.on('error', errorHandler);

      // 空闲超时检测
      timeoutTimer = setTimeout(() => {
        errorHandler(new Error('数据传输超时,疑似网络中断'));
      }, CONFIG.timeout);

      // 监听数据块写入
      rs.on('data', (chunk) => {
        downloaded += chunk.length;
        clearTimeout(timeoutTimer);
        timeoutTimer = setTimeout(() => {
          errorHandler(new Error('数据传输超时,疑似网络中断'));
        }, CONFIG.timeout);
        const progress = ((downloaded / totalSize) * 100).toFixed(2);
        console.log(`下载进度: ${progress}%`);
      });

      // 等待当前段传输完成
      await new Promise((resolve, reject) => {
        ws.on('finish', resolve);
        ws.on('error', reject);
        rs.pipe(ws);
      });

      clearTimeout(timeoutTimer);
      retryCount = 0; // 传输成功重置重试计数

    } catch (e) {
      clearTimeout(timeoutTimer);
      // 非网络错误直接抛出
      if (!NETWORK_ERRORS.has(e.code) && !e.message.includes('超时')) {
        throw e;
      }
      retryCount++;
      if (retryCount > CONFIG.maxRetry) {
        throw new Error(`超过最大重试次数${CONFIG.maxRetry},下载失败`);
      }
      // 指数退避等待
      const waitTime = Math.min(1000 * Math.pow(2, retryCount - 1), 16000);
      console.log(`网络异常,${waitTime/1000}秒后重试,错误信息: ${e.message}`);
      await new Promise(resolve => setTimeout(resolve, waitTime));
      // 重新统计本地实际写入大小,避免最后一块数据未写完整
      downloaded = await getLocalFileSize(CONFIG.localPath);
    }
  }

  // 最终完整性校验
  const finalSize = await getLocalFileSize(CONFIG.localPath);
  if (finalSize !== totalSize) {
    await fs.unlink(CONFIG.localPath);
    throw new Error("文件大小校验失败,已删除损坏文件,请重新运行下载");
  }
  console.log("下载完成,文件校验通过");
}

download().catch(err => {
  console.error("下载失败:", err);
  process.exit(1);
});
核心逻辑说明
  • 断点续传:每次发起GetObject请求时携带Range: bytes=${已下载字节数}-请求头,S3仅返回指定位置之后的文件内容,本地文件用追加模式写入,不会覆盖已下载的部分
  • 异常捕获:同时绑定可读流、可写流的error事件,额外加了空闲超时检测——如果连续30秒没有收到任何数据块,直接判定为网络中断,主动中断当前连接进入重试流程,不会出现无响应卡死的情况
  • 自动续传:捕获到网络类错误(超时、连接重置、DNS解析失败等)后,采用指数退避策略等待重试,重试前先重新统计本地文件实际写入的字节数(避免流中断时最后一个数据块未写完整导致文件损坏),从断点位置重新发起请求
  • 完整性校验:下载完成后对比本地文件大小和S3对象的ContentLength,大小不一致则删除损坏文件,避免解压zip时报错
配置调整说明

可根据实际场景修改代码开头CONFIG对象的参数:

  • timeout:数据传输空闲超时时间,弱网环境可以适当调大到60000(60秒)
  • maxRetry:最大重试次数,网络环境差可以调到20以上
  • 本地路径、S3密钥、桶名、对象Key按实际信息填写即可

如果需要Python版本实现,逻辑完全一致:用boto3获取对象元数据,循环发起带Range头的下载请求,追加写入本地文件,捕获requests库抛出的网络异常做退避重试即可。

内容的提问来源于stack exchange,提问作者Guy-dev

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.28 08:09:26