You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

NodeJS Cheerio爬虫无法获取搜索文本对应标签名的技术求助

Fixing Your Cheerio Crawler to Show Target Text's HTML Tag

Hey there! Let's get your crawler to not only detect when your target text exists on a page, but also tell you exactly which HTML tags contain that text. The issue with your current searchForWord function is that it checks the entire body's raw HTML/text as a single block—so it can't pinpoint individual elements. Here's how to fix it:

Step 1: Rewrite the searchForWord Function

Instead of checking the whole body at once, we'll traverse every relevant element on the page, check if its text contains your target word, and collect the corresponding tag names. We'll also avoid duplicate tags and skip non-content elements like <script> or <style> that don't hold visible text.

function searchForWord($, word) {
  const targetWord = word.toLowerCase();
  const foundTags = [];

  // Traverse all elements in the body except non-content tags
  $('body *:not(script):not(style):not(noscript)').each((index, element) => {
    const $element = $(element);
    const elementText = $element.text().trim().toLowerCase();

    // Check if this element directly contains the target word
    if (elementText.includes(targetWord)) {
      // Optional: Uncomment below if you only want the innermost tag containing the text
      // const parentText = $element.parent().text().trim().toLowerCase();
      // if (!parentText.includes(targetWord)) {
        foundTags.push($element.prop('tagName'));
      // }
    }
  });

  // Remove duplicate tag names
  return [...new Set(foundTags)];
}

Step 2: Update the visitPage Function to Use the New Results

Now we'll adjust how we handle the result from searchForWord—instead of a boolean, it returns an array of tag names. We'll loop through that array to print your desired output format:

function visitPage(url, callback) {
  // Add page to our set
  pagesVisited[url] = true;
  numPagesVisited++;

  // Make the request
  request(url, function(error, response, body) {
    console.log("***************************")
    console.log(" Visiting page: " + url + '\n');
    if(response.statusCode !== 200) {
      callback();
      return;
    }

    // Parse the document body
    var $ = cheerio.load(body);
    var foundTags = searchForWord($, SEARCH_WORD);
    
    if(foundTags.length > 0) {
      // Loop through each found tag and print the formatted result
      foundTags.forEach(tag => {
        console.log(`text "${SEARCH_WORD}" found on ${url} of "TAGNAME: ${tag}"`);
      });
      collectInternalLinks($);
      callback();
    } else {
      collectInternalLinks($);
      callback();
    }
  });
}

Key Notes:

  • Avoiding Duplicates: The [...new Set(foundTags)] line removes repeated tag names (e.g., if the text appears in multiple <p> tags, it only shows P once).
  • Innermost Tag Option: The commented-out parent check ensures we only record the deepest element containing the text (so if a <span> inside a <div> has the text, we only get SPAN instead of both SPAN and DIV). Remove the comment if you want this behavior.
  • Case Insensitivity: We convert both the target word and element text to lowercase to match regardless of capitalization, which keeps your original functionality intact.

Full Modified Code

Here's the complete updated index.js with all changes applied:

var request = require('request');
var cheerio = require('cheerio');
var URL = require('url-parse');
var START_URL = "https://www.mytravelexp.com/";
var SEARCH_WORD ="Pack your travel essentials";
var MAX_PAGES_TO_VISIT = 20;
var pagesVisited = {};
var numPagesVisited = 0;
var pagesToVisit = [];
var url = new URL(START_URL);
var baseUrl = url.protocol + "//" + url.hostname;
pagesToVisit.push(START_URL);
crawl();

function crawl() {
  if(numPagesVisited >= MAX_PAGES_TO_VISIT) {
    console.log("Reached max limit of number of pages to visit.");
    return;
  }
  var nextPage = pagesToVisit.pop();
  if (nextPage in pagesVisited) {
    // We've already visited this page, so repeat the crawl
    crawl();
  } else {
    // New page we haven't visited
    visitPage(nextPage, crawl);
  }
}

function visitPage(url, callback) {
  // Add page to our set
  pagesVisited[url] = true;
  numPagesVisited++;

  // Make the request
  request(url, function(error, response, body) {
    console.log("***************************")
    console.log(" Visiting page: " + url + '\n');
    if(response.statusCode !== 200) {
      callback();
      return;
    }

    // Parse the document body
    var $ = cheerio.load(body);
    var foundTags = searchForWord($, SEARCH_WORD);
    
    if(foundTags.length > 0) {
      foundTags.forEach(tag => {
        console.log(`text "${SEARCH_WORD}" found on ${url} of "TAGNAME: ${tag}"`);
      });
      collectInternalLinks($);
      callback();
    } else {
      collectInternalLinks($);
      callback();
    }
  });
}

function searchForWord($, word) {
  const targetWord = word.toLowerCase();
  const foundTags = [];

  $('body *:not(script):not(style):not(noscript)').each((index, element) => {
    const $element = $(element);
    const elementText = $element.text().trim().toLowerCase();

    if (elementText.includes(targetWord)) {
      // Optional: Uncomment to only track innermost tags
      // const parentText = $element.parent().text().trim().toLowerCase();
      // if (!parentText.includes(targetWord)) {
        foundTags.push($element.prop('tagName'));
      // }
    }
  });

  return [...new Set(foundTags)];
}

function collectInternalLinks($) {
  var relativeLinks = $("a[href^='/']");
  relativeLinks.each(function() {
    pagesToVisit.push(baseUrl + $(this).attr('href'));
  });
  var absoluteLinks = $("a[href^='http']");
  absoluteLinks.each(function() {
    pagesToVisit.push($(this).attr('href'));
  });
}

内容的提问来源于stack exchange,提问作者Bindu Malik

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.04.29 19:44:06