You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Node.js使用node-downloader-helper下载PDF报408响应错误

问题背景

需要基于Node.js实现PDF批量下载功能,使用node-downloader-helper模块编写了如下downloadPDF函数,该函数会被多次调用以下载完整的PDF列表:

function downloadPDF(url, dirname, index) {
    const options = {
        method: "GET", // Request Method Verb
        headers: {}, // Custom HTTP Header ex: Authorization, User-Agent
        fileName: `document${index}.pdf`, // Custom filename when saved
        forceResume: false, // If the server does not return the "accept-ranges" header, can be force if it does support it
        removeOnStop: true, // remove the file when is stopped (default:true)
        removeOnFail: true, // remove the file when fail (default:true)
        httpRequestOptions: {}, // Override the http request options
        httpsRequestOptions: {
            timeout: 10000, // Timeout in milliseconds
        }, // Override the https request options, ex: to add SSL Certs
        timeout: 10000, // Timeout in ms, 0 to disable (default:0)
    };

    const dl = new DownloaderHelper(encodeURI(url), dirname, options);

    dl.on('end', () => console.log('Download Completed'));
    dl.on('error', (err) => console.log('Download Failed', err));
    dl.start().catch(err => console.error(err));
}

运行程序时返回如下错误,部分场景下还会直接导致服务器完全崩溃。初步判断该问题与超时有关,但尝试多种调整方法均未解决:

Error: Response status was 408
    at ClientRequest.<anonymous> (/home/ubuntu/CPE-Colle/node_modules/node-downloader-helper/dist/index.js:1:7055)
    at Object.onceWrapper (node:events:642:26)
    at ClientRequest.emit (node:events:527:28)
    at HTTPParser.parserOnIncomingClient [as onIncoming] (node:_http_client:631:27)
    at HTTPParser.parserOnHeadersComplete (node:_http_common:117:17)
    at TLSSocket.socketOnData (node:_http_client:494:22)
    at TLSSocket.emit (node:events:527:28)
    at addChunk (node:internal/streams/readable:324:12)
    at readableAddChunk (node:internal/streams/readable:297:9)
    at Readable.push (node:internal/streams/readable:234:10) {
  status: 408,
  body: ''
}
问题根因

408状态码是服务端返回的「请求超时」响应,不是本地设置的客户端超时主动触发的,所以单纯调大客户端超时时间无法解决问题,具体触发原因有4点:

  • 无并发控制:批量调用downloadPDF时瞬间发起大量请求,目标服务器的限流策略会直接挂起多余请求,超时后返回408;请求堆积占满内存时就会直接打崩服务进程
  • 请求头缺失:默认请求没有携带常规浏览器的User-Agent等标识头,多数站点的WAF防护会把这类请求判定为恶意爬虫,故意挂起请求直到超时返回408
  • 超时配置冲突:同时在httpsRequestOptions和外层options里配置了10s超时,部分版本的node-downloader-helper会出现超时事件重复触发、错误未被正确捕获的问题,未捕获的异常会直接导致进程退出
  • URL编码错误:手动对整个URL调用encodeURI,会把协议头、路径里本来合规的特殊字符错误转义,目标服务器识别不了异常格式的URL,就会挂起请求直到超时
修复方案

按以下步骤调整即可解决问题:

  • 补全合法请求头:在headers里加上常规浏览器的User-Agent,如果目标站点有防盗链,再补充对应域名的Referer头,避免被WAF拦截
  • 移除重复超时配置:删掉httpsRequestOptions里的timeout设置,只保留外层options的timeout参数,可适当将超时调整到30s适配体积较大的PDF资源
  • 移除错误的全量URL编码:删掉手动调用的encodeURI,node-downloader-helper内部已经做了合规的URL编码处理,手动全量编码反而会导致URL格式错误
  • 增加并发控制:批量下载时限制同时进行的下载任务数在3-5个,不要一次性发起几十上百个请求
  • 增加失败重试逻辑:遇到408等临时错误时自动延迟重试2-3次,同时补全所有异常分支的捕获逻辑,避免未处理的错误打崩进程

修复后的完整可运行代码如下:

const { DownloaderHelper } = require('node-downloader-helper');

// 简单并发控制队列,限制同时执行的任务数
class TaskQueue {
  constructor(concurrency = 3) {
    this.concurrency = concurrency;
    this.running = 0;
    this.queue = [];
  }
  addTask(task) {
    return new Promise((resolve, reject) => {
      this.queue.push({ task, resolve, reject });
      this.run();
    })
  }
  run() {
    if (this.running >= this.concurrency || this.queue.length === 0) return;
    const { task, resolve, reject } = this.queue.shift();
    this.running++;
    task()
      .then(resolve)
      .catch(reject)
      .finally(() => {
        this.running--;
        this.run();
      })
  }
}
// 初始化下载队列,最多同时下载3个文件,可根据目标站点性能调整
const downloadQueue = new TaskQueue(3);

async function downloadPDF(url, dirname, index, retryCount = 2) {
  return downloadQueue.addTask(() => new Promise((resolve, reject) => {
    const options = {
      method: "GET",
      headers: {
        // 替换为当前主流浏览器的UA即可
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
      },
      fileName: `document${index}.pdf`,
      forceResume: false,
      removeOnStop: true,
      removeOnFail: true,
      httpRequestOptions: {},
      httpsRequestOptions: {},
      timeout: 30000, // 仅保留这一处超时配置
      retry: { maxRetries: 1, delay: 1000 } // 开启库自带的单次请求重试
    };

    const dl = new DownloaderHelper(url, dirname, options);
    dl.on('end', () => {
      console.log(`第${index}个PDF下载完成`);
      resolve();
    });
    dl.on('error', (err) => {
      // 408错误且还有重试次数时,延迟2s重试
      if ((err.status === 408 || err.code === 'ETIMEDOUT') && retryCount > 0) {
        console.log(`第${index}个PDF下载超时,剩余重试次数${retryCount},2s后重试`);
        setTimeout(() => {
          downloadPDF(url, dirname, index, retryCount - 1).then(resolve).catch(reject);
        }, 2000);
        return;
      }
      console.log(`第${index}个PDF下载失败`, err.message);
      reject(err);
    });
    dl.start().catch(err => {
      if (retryCount > 0) {
        setTimeout(() => {
          downloadPDF(url, dirname, index, retryCount - 1).then(resolve).catch(reject);
        }, 2000);
        return;
      }
      reject(err);
    });
  }))
}

// 批量下载调用示例
async function batchDownload(pdfUrlList, saveDir) {
  // 用allSettled避免单个任务失败中断整个批量流程
  const tasks = pdfUrlList.map((url, idx) => downloadPDF(url, saveDir, idx));
  const results = await Promise.allSettled(tasks);
  const successCount = results.filter(item => item.status === 'fulfilled').length;
  console.log(`全部下载任务执行完毕,成功${successCount}个,失败${results.length - successCount}个`);
}

额外注意事项

  • 如果目标站点防盗链严格,可在headers中添加Referer字段,值为目标站点的根域名即可
  • 并发数不要设置过高,普通公开站点3-5并发足够,若目标站点服务器性能较差,可降到2
  • 只有当URL路径中存在中文等非ASCII特殊字符时,才需要单独对路径部分做编码,不要对整个URL调用encodeURI

内容的提问来源于stack exchange,提问作者MaxBrt18

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.02 23:12:34