Node.js使用node-downloader-helper下载PDF报408响应错误
问题背景
需要基于Node.js实现PDF批量下载功能,使用node-downloader-helper模块编写了如下downloadPDF函数,该函数会被多次调用以下载完整的PDF列表:
function downloadPDF(url, dirname, index) { const options = { method: "GET", // Request Method Verb headers: {}, // Custom HTTP Header ex: Authorization, User-Agent fileName: `document${index}.pdf`, // Custom filename when saved forceResume: false, // If the server does not return the "accept-ranges" header, can be force if it does support it removeOnStop: true, // remove the file when is stopped (default:true) removeOnFail: true, // remove the file when fail (default:true) httpRequestOptions: {}, // Override the http request options httpsRequestOptions: { timeout: 10000, // Timeout in milliseconds }, // Override the https request options, ex: to add SSL Certs timeout: 10000, // Timeout in ms, 0 to disable (default:0) }; const dl = new DownloaderHelper(encodeURI(url), dirname, options); dl.on('end', () => console.log('Download Completed')); dl.on('error', (err) => console.log('Download Failed', err)); dl.start().catch(err => console.error(err)); }
运行程序时返回如下错误,部分场景下还会直接导致服务器完全崩溃。初步判断该问题与超时有关,但尝试多种调整方法均未解决:
Error: Response status was 408 at ClientRequest.<anonymous> (/home/ubuntu/CPE-Colle/node_modules/node-downloader-helper/dist/index.js:1:7055) at Object.onceWrapper (node:events:642:26) at ClientRequest.emit (node:events:527:28) at HTTPParser.parserOnIncomingClient [as onIncoming] (node:_http_client:631:27) at HTTPParser.parserOnHeadersComplete (node:_http_common:117:17) at TLSSocket.socketOnData (node:_http_client:494:22) at TLSSocket.emit (node:events:527:28) at addChunk (node:internal/streams/readable:324:12) at readableAddChunk (node:internal/streams/readable:297:9) at Readable.push (node:internal/streams/readable:234:10) { status: 408, body: '' }
问题根因
408状态码是服务端返回的「请求超时」响应,不是本地设置的客户端超时主动触发的,所以单纯调大客户端超时时间无法解决问题,具体触发原因有4点:
- 无并发控制:批量调用
downloadPDF时瞬间发起大量请求,目标服务器的限流策略会直接挂起多余请求,超时后返回408;请求堆积占满内存时就会直接打崩服务进程 - 请求头缺失:默认请求没有携带常规浏览器的
User-Agent等标识头,多数站点的WAF防护会把这类请求判定为恶意爬虫,故意挂起请求直到超时返回408 - 超时配置冲突:同时在
httpsRequestOptions和外层options里配置了10s超时,部分版本的node-downloader-helper会出现超时事件重复触发、错误未被正确捕获的问题,未捕获的异常会直接导致进程退出 - URL编码错误:手动对整个URL调用
encodeURI,会把协议头、路径里本来合规的特殊字符错误转义,目标服务器识别不了异常格式的URL,就会挂起请求直到超时
修复方案
按以下步骤调整即可解决问题:
- 补全合法请求头:在headers里加上常规浏览器的
User-Agent,如果目标站点有防盗链,再补充对应域名的Referer头,避免被WAF拦截 - 移除重复超时配置:删掉
httpsRequestOptions里的timeout设置,只保留外层options的timeout参数,可适当将超时调整到30s适配体积较大的PDF资源 - 移除错误的全量URL编码:删掉手动调用的
encodeURI,node-downloader-helper内部已经做了合规的URL编码处理,手动全量编码反而会导致URL格式错误 - 增加并发控制:批量下载时限制同时进行的下载任务数在3-5个,不要一次性发起几十上百个请求
- 增加失败重试逻辑:遇到408等临时错误时自动延迟重试2-3次,同时补全所有异常分支的捕获逻辑,避免未处理的错误打崩进程
修复后的完整可运行代码如下:
const { DownloaderHelper } = require('node-downloader-helper'); // 简单并发控制队列,限制同时执行的任务数 class TaskQueue { constructor(concurrency = 3) { this.concurrency = concurrency; this.running = 0; this.queue = []; } addTask(task) { return new Promise((resolve, reject) => { this.queue.push({ task, resolve, reject }); this.run(); }) } run() { if (this.running >= this.concurrency || this.queue.length === 0) return; const { task, resolve, reject } = this.queue.shift(); this.running++; task() .then(resolve) .catch(reject) .finally(() => { this.running--; this.run(); }) } } // 初始化下载队列,最多同时下载3个文件,可根据目标站点性能调整 const downloadQueue = new TaskQueue(3); async function downloadPDF(url, dirname, index, retryCount = 2) { return downloadQueue.addTask(() => new Promise((resolve, reject) => { const options = { method: "GET", headers: { // 替换为当前主流浏览器的UA即可 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36" }, fileName: `document${index}.pdf`, forceResume: false, removeOnStop: true, removeOnFail: true, httpRequestOptions: {}, httpsRequestOptions: {}, timeout: 30000, // 仅保留这一处超时配置 retry: { maxRetries: 1, delay: 1000 } // 开启库自带的单次请求重试 }; const dl = new DownloaderHelper(url, dirname, options); dl.on('end', () => { console.log(`第${index}个PDF下载完成`); resolve(); }); dl.on('error', (err) => { // 408错误且还有重试次数时,延迟2s重试 if ((err.status === 408 || err.code === 'ETIMEDOUT') && retryCount > 0) { console.log(`第${index}个PDF下载超时,剩余重试次数${retryCount},2s后重试`); setTimeout(() => { downloadPDF(url, dirname, index, retryCount - 1).then(resolve).catch(reject); }, 2000); return; } console.log(`第${index}个PDF下载失败`, err.message); reject(err); }); dl.start().catch(err => { if (retryCount > 0) { setTimeout(() => { downloadPDF(url, dirname, index, retryCount - 1).then(resolve).catch(reject); }, 2000); return; } reject(err); }); })) } // 批量下载调用示例 async function batchDownload(pdfUrlList, saveDir) { // 用allSettled避免单个任务失败中断整个批量流程 const tasks = pdfUrlList.map((url, idx) => downloadPDF(url, saveDir, idx)); const results = await Promise.allSettled(tasks); const successCount = results.filter(item => item.status === 'fulfilled').length; console.log(`全部下载任务执行完毕,成功${successCount}个,失败${results.length - successCount}个`); }
额外注意事项
- 如果目标站点防盗链严格,可在headers中添加
Referer字段,值为目标站点的根域名即可 - 并发数不要设置过高,普通公开站点3-5并发足够,若目标站点服务器性能较差,可降到2
- 只有当URL路径中存在中文等非ASCII特殊字符时,才需要单独对路径部分做编码,不要对整个URL调用
encodeURI
内容的提问来源于stack exchange,提问作者MaxBrt18
相关产品推荐
相关产品推荐

