You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用Puppeteer和Cheerio提取动态无类名<ul>的文本?

动态
    元素的文本提取方案(Puppeteer/Cheerio)

核心思路

放弃不可靠的nth-child定位,利用**

标题与对应
    的相邻兄弟关系**来定位——这是页面结构中稳定的关联逻辑,不受元素数量动态变化的影响。


    方法1:纯Puppeteer实现

    直接在浏览器上下文里遍历标题,关联对应的列表:

    const scrapeUl = async (page, singleSku) => {
      // 保留原有的页面跳转、搜索逻辑
      await page.goto("https://myweb.com", { waitUntil: "networkidle2" });
      await page.waitForXPath('//*[@id="parent-container-id"]/section/header/div/div/div[1]/form/div[1]/div/div');
      let [Input] = await page.$x('//*[@id="parent-container-id"]/section/header/div/div/div[1]/form/div[1]/div/div/input');
      await Input.type(singleSku);
      await Promise.all([
        page.waitForNavigation({waitUntil: "domcontentloaded"}),
        page.keyboard.press("Enter"),
      ]);
    
      // 等待目标容器加载完成
      await page.waitForSelector('div.card-section > div.right.text--pull');
    
      // 提取所有标题对应的<ul>内容,返回键值对对象
      const allSections = await page.evaluate(() => {
        const container = document.querySelector('div.card-section > div.right.text--pull');
        const sections = {};
        
        container.querySelectorAll('h6').forEach(h6 => {
          const title = h6.textContent.trim();
          const nextUl = h6.nextElementSibling;
          
          if (nextUl && nextUl.tagName === 'UL') {
            // 若需要拆分每个li为数组,替换为下面一行:
            // sections[title] = Array.from(nextUl.querySelectorAll('li')).map(li => li.textContent.trim());
            sections[title] = nextUl.innerText.trim();
          }
        });
        
        return sections;
      });
    
      // 使用示例:获取Features对应的内容
      console.log(allSections['Features']);
      // 获取Includes对应的内容
      console.log(allSections['Includes']);
      
      return allSections;
    };
    

    方法2:Puppeteer + Cheerio结合

    先抓取目标区域HTML,再用Cheerio灵活解析:

    const cheerio = require('cheerio');
    
    const scrapeUl = async (page, singleSku) => {
      // 保留原有的页面跳转、搜索逻辑
      await page.goto("https://myweb.com", { waitUntil: "networkidle2" });
      await page.waitForXPath('//*[@id="parent-container-id"]/section/header/div/div/div[1]/form/div[1]/div/div');
      let [Input] = await page.$x('//*[@id="parent-container-id"]/section/header/div/div/div[1]/form/div[1]/div/div/input');
      await Input.type(singleSku);
      await Promise.all([
        page.waitForNavigation({waitUntil: "domcontentloaded"}),
        page.keyboard.press("Enter"),
      ]);
    
      // 获取目标区域的HTML
      const containerHtml = await page.$eval('div.card-section > div.right.text--pull', el => el.outerHTML);
      const $ = cheerio.load(containerHtml);
    
      // 提取所有标题对应的<ul>内容
      const sections = {};
      $('h6').each((i, el) => {
        const title = $(el).text().trim();
        const $ul = $(el).next('ul');
        
        if ($ul.length) {
          // 若需要拆分每个li为数组,替换为下面一行:
          // sections[title] = $ul.find('li').map((i, li) => $(li).text().trim()).get();
          sections[title] = $ul.text().trim();
        }
      });
    
      // 单独提取指定标题的内容(比如Includes)
      const includesText = $('h6:contains("Includes")').next('ul').text().trim();
      console.log(includesText);
      
      return sections;
    };
    

    关键注意点

    1. 等待DOM加载:必须用page.waitForSelector确保目标容器渲染完成,避免提取空内容。
    2. 文本去空格:用trim()处理标题文本,避免因空格导致匹配失败。
    3. 容错处理:判断相邻元素是否为<ul>,避免页面结构异常时出错。

    内容的提问来源于stack exchange,提问作者Pankaj Mishra

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.17 13:45:31