You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用Puppeteer多页爬取时第二页出现TimeoutError问题求助

问题分析与修复方案

核心问题点

  • 爬取函数调用错误:scrapeEpisode.concat(page)是完全错误的写法,concat是数组方法,这里应该直接调用await scrapeEpisode(page),这个错误会导致程序无法执行第二页爬取逻辑,进而卡住超时。
  • 分页参数生成错误:初始nextPageNumber=1,且直接在原URL后追加&spage=xxx,会生成带重复参数的URL(比如...&spage=2&spage=3),导致页面加载异常或返回无效内容,触发超时。
  • 爬取函数未返回数据:scrapeEpisode仅打印数据,没有返回结果,无法收集多页数据。
  • 页面加载策略不合理:默认的等待规则和超时设置不匹配目标网站加载速度,容易触发超时。

修复后的完整代码

const puppeteer = require("puppeteer-extra")
const StealthPlugin = require("puppeteer-extra-plugin-stealth")

puppeteer.use(StealthPlugin());

// 修改爬取函数,返回当前页数据用于收集
async function scrapeEpisode(page){
  const EpisodesOnPage = await page.evaluate(() =>
    Array.from(document.querySelectorAll("div.wr-subject")).map(compact => ({
      title: compact.innerText.trim(),
      link: compact.querySelector("a").href
    }))
  );
  console.log(`当前页爬取到 ${EpisodesOnPage.length} 条数据`);
  return EpisodesOnPage;
}

(async () => {
  const browser = await puppeteer.launch({
    headless: false,
    targetFilter: (target) => target.type() !== "other",
    // 添加防反爬配置
    defaultViewport: null,
    args: ['--no-sandbox', '--disable-setuid-sandbox', '--user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36']
  });

  // 重构分页爬取逻辑
  const callURL = async (url, browser) => {
    const page = await browser.newPage();
    let allEpisodes = [];
    let currentPageUrl = url;

    while (true) {
      console.log(`正在访问: ${currentPageUrl}`);
      try {
        // 调整加载策略,延长超时时间
        await page.goto(currentPageUrl, {
          timeout: 60000, // 延长至60秒超时
          waitUntil: 'domcontentloaded', // 仅等待DOM加载完成,无需等所有资源
        });

        // 等待目标元素,设置超时
        await page.waitForSelector('#viewcomment', { timeout: 30000 });
        
        // 收集当前页数据
        const pageData = await scrapeEpisode(page);
        allEpisodes = allEpisodes.concat(pageData);

        // 获取下一页真实链接(避免手动拼接参数出错)
        const nextPageBtn = await page.$('a:has(.pg_next)'); // 可根据网站实际按钮选择器调整
        if (!nextPageBtn) {
          console.log('已无更多页面,停止爬取');
          break;
        }
        currentPageUrl = await page.evaluate(el => el.href, nextPageBtn);

      } catch (error) {
        console.error(`处理页面 ${currentPageUrl} 时出错:`, error);
        break;
      }
    }

    await page.close();
    return allEpisodes;
  }

  const initialUrl = 'https://booktoki460.com/novel/222?page=2&book=%EC%9D%BC%EB%B0%98%EC%86%8C%EC%84%A4';
  const allEpisodes = await callURL(initialUrl, browser);
  console.log(`所有页爬取完成,共 ${allEpisodes.length} 条数据`);
  console.log(allEpisodes);
  
  await browser.close();
})();

关键修复说明

  1. 修正函数调用:把错误的scrapeEpisode.concat(page)改为await scrapeEpisode(page),并让函数返回数据,实现多页数据收集。
  2. 优化分页逻辑:不再手动拼接URL参数,直接从页面获取下一页按钮的真实链接,避免参数格式错误导致的页面加载异常。
  3. 调整加载策略:
    • 延长超时时间到60秒,适配网站加载速度;
    • 使用domcontentloaded替代默认的load,无需等待图片、广告等非必要资源,减少超时概率;
    • 添加自定义User-Agent,降低被反爬识别的概率。
  4. 完善错误处理:添加try-catch捕获页面处理中的错误,避免程序直接崩溃。
  5. 资源清理:爬取结束后关闭页面和浏览器,避免资源泄漏。

内容的提问来源于stack exchange,提问作者Oli Ver

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.18 13:15:04