NodeJS Cheerio爬虫无法获取搜索文本对应标签名的技术求助
Hey there! Let's get your crawler to not only detect when your target text exists on a page, but also tell you exactly which HTML tags contain that text. The issue with your current searchForWord function is that it checks the entire body's raw HTML/text as a single block—so it can't pinpoint individual elements. Here's how to fix it:
Step 1: Rewrite the searchForWord Function
Instead of checking the whole body at once, we'll traverse every relevant element on the page, check if its text contains your target word, and collect the corresponding tag names. We'll also avoid duplicate tags and skip non-content elements like <script> or <style> that don't hold visible text.
function searchForWord($, word) { const targetWord = word.toLowerCase(); const foundTags = []; // Traverse all elements in the body except non-content tags $('body *:not(script):not(style):not(noscript)').each((index, element) => { const $element = $(element); const elementText = $element.text().trim().toLowerCase(); // Check if this element directly contains the target word if (elementText.includes(targetWord)) { // Optional: Uncomment below if you only want the innermost tag containing the text // const parentText = $element.parent().text().trim().toLowerCase(); // if (!parentText.includes(targetWord)) { foundTags.push($element.prop('tagName')); // } } }); // Remove duplicate tag names return [...new Set(foundTags)]; }
Step 2: Update the visitPage Function to Use the New Results
Now we'll adjust how we handle the result from searchForWord—instead of a boolean, it returns an array of tag names. We'll loop through that array to print your desired output format:
function visitPage(url, callback) { // Add page to our set pagesVisited[url] = true; numPagesVisited++; // Make the request request(url, function(error, response, body) { console.log("***************************") console.log(" Visiting page: " + url + '\n'); if(response.statusCode !== 200) { callback(); return; } // Parse the document body var $ = cheerio.load(body); var foundTags = searchForWord($, SEARCH_WORD); if(foundTags.length > 0) { // Loop through each found tag and print the formatted result foundTags.forEach(tag => { console.log(`text "${SEARCH_WORD}" found on ${url} of "TAGNAME: ${tag}"`); }); collectInternalLinks($); callback(); } else { collectInternalLinks($); callback(); } }); }
Key Notes:
- Avoiding Duplicates: The
[...new Set(foundTags)]line removes repeated tag names (e.g., if the text appears in multiple<p>tags, it only showsPonce). - Innermost Tag Option: The commented-out parent check ensures we only record the deepest element containing the text (so if a
<span>inside a<div>has the text, we only getSPANinstead of bothSPANandDIV). Remove the comment if you want this behavior. - Case Insensitivity: We convert both the target word and element text to lowercase to match regardless of capitalization, which keeps your original functionality intact.
Full Modified Code
Here's the complete updated index.js with all changes applied:
var request = require('request'); var cheerio = require('cheerio'); var URL = require('url-parse'); var START_URL = "https://www.mytravelexp.com/"; var SEARCH_WORD ="Pack your travel essentials"; var MAX_PAGES_TO_VISIT = 20; var pagesVisited = {}; var numPagesVisited = 0; var pagesToVisit = []; var url = new URL(START_URL); var baseUrl = url.protocol + "//" + url.hostname; pagesToVisit.push(START_URL); crawl(); function crawl() { if(numPagesVisited >= MAX_PAGES_TO_VISIT) { console.log("Reached max limit of number of pages to visit."); return; } var nextPage = pagesToVisit.pop(); if (nextPage in pagesVisited) { // We've already visited this page, so repeat the crawl crawl(); } else { // New page we haven't visited visitPage(nextPage, crawl); } } function visitPage(url, callback) { // Add page to our set pagesVisited[url] = true; numPagesVisited++; // Make the request request(url, function(error, response, body) { console.log("***************************") console.log(" Visiting page: " + url + '\n'); if(response.statusCode !== 200) { callback(); return; } // Parse the document body var $ = cheerio.load(body); var foundTags = searchForWord($, SEARCH_WORD); if(foundTags.length > 0) { foundTags.forEach(tag => { console.log(`text "${SEARCH_WORD}" found on ${url} of "TAGNAME: ${tag}"`); }); collectInternalLinks($); callback(); } else { collectInternalLinks($); callback(); } }); } function searchForWord($, word) { const targetWord = word.toLowerCase(); const foundTags = []; $('body *:not(script):not(style):not(noscript)').each((index, element) => { const $element = $(element); const elementText = $element.text().trim().toLowerCase(); if (elementText.includes(targetWord)) { // Optional: Uncomment to only track innermost tags // const parentText = $element.parent().text().trim().toLowerCase(); // if (!parentText.includes(targetWord)) { foundTags.push($element.prop('tagName')); // } } }); return [...new Set(foundTags)]; } function collectInternalLinks($) { var relativeLinks = $("a[href^='/']"); relativeLinks.each(function() { pagesToVisit.push(baseUrl + $(this).attr('href')); }); var absoluteLinks = $("a[href^='http']"); absoluteLinks.each(function() { pagesToVisit.push($(this).attr('href')); }); }
内容的提问来源于stack exchange,提问作者Bindu Malik

