如何使用Node.js获取页面全部字体并批量爬取站点页面
Hey there! Let's tackle your two Node.js tasks—scraping all fonts from pages and running your existing site crawler on a server. I'll break this down step by step, building on the code snippet you shared.
Your existing crawler structure is solid—we just need to add logic to detect fonts from every crawled page. Fonts can come from inline styles, internal CSS, external stylesheets, and @font-face rules, so we'll cover all these cases.
Here's the updated, full script with font extraction integrated:
const request = require('request'); const cheerio = require('cheerio'); const URL = require('url-parse'); const jsdom = require("jsdom"); const { JSDOM } = jsdom; const START_URL = "http://balneol.com/"; const MAX_PAGES_TO_VISIT = 100000; const pagesVisited = {}; let numPagesVisited = 0; const pagesToVisit = []; const detectedFonts = new Set(); // Use Set to avoid duplicate font names const url = new URL(START_URL); const baseUrl = url.protocol + "//" + url.hostname; pagesToVisit.push(START_URL); crawl(); function crawl() { if (numPagesVisited >= MAX_PAGES_TO_VISIT) { console.log("\n=== Crawl Complete ==="); console.log(`Total pages visited: ${numPagesVisited}`); console.log("Detected fonts:", Array.from(detectedFonts).sort()); return; } const nextPage = pagesToVisit.pop(); if (nextPage in pagesVisited) { crawl(); // Skip already visited pages } else { visitPage(nextPage, crawl); } } function visitPage(url, callback) { pagesVisited[url] = true; numPagesVisited++; console.log(`[${numPagesVisited}/${MAX_PAGES_TO_VISIT}] Visiting: ${url}`); request( { url: url, headers: { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36' } }, function(error, response, body) { if (error) { console.log(`Error accessing ${url}: ${error.message}`); callback(); return; } if (response.statusCode !== 200) { console.log(`Non-200 status for ${url}: ${response.statusCode}`); callback(); return; } extractFonts(body, url); // Extract fonts from the page collectInternalLinks(cheerio.load(body)); // Gather more pages to crawl callback(); } ); } function extractFonts(html, pageUrl) { const $ = cheerio.load(html); // 1. Extract fonts from inline styles and element CSS $('*').each((_, elem) => { const fontFamily = $(elem).css('font-family'); if (fontFamily) { const cleanFonts = fontFamily.split(',') .map(font => font.trim().replace(/['"]/g, '').toLowerCase()) .filter(Boolean); cleanFonts.forEach(font => detectedFonts.add(font)); } }); // 2. Extract fonts from internal <style> tags (including @font-face) $('style').each((_, style) => { const cssText = $(style).text(); // Match @font-face font families const fontFaceRegex = /@font-face\s*{[^}]*font-family:\s*['"]?([^'";]+)['"]?[^}]*}/gi; let match; while ((match = fontFaceRegex.exec(cssText)) !== null) { detectedFonts.add(match[1].trim().toLowerCase()); } }); // 3. Use JSDOM to get computed fonts (more accurate for inherited styles) const dom = new JSDOM(html, { url: pageUrl }); const allElements = dom.window.document.querySelectorAll('*'); allElements.forEach(elem => { const computedStyle = dom.window.getComputedStyle(elem); const fontFamily = computedStyle.fontFamily; if (fontFamily) { const cleanFonts = fontFamily.split(',') .map(font => font.trim().replace(/['"]/g, '').toLowerCase()) .filter(Boolean); cleanFonts.forEach(font => detectedFonts.add(font)); } }); // 4. Extract fonts from external stylesheets $('link[rel="stylesheet"]').each((_, link) => { const cssUrl = new URL($(link).attr('href'), baseUrl).href; request(cssUrl, (err, res, cssBody) => { if (!err && res.statusCode === 200) { const fontFaceRegex = /@font-face\s*{[^}]*font-family:\s*['"]?([^'";]+)['"]?[^}]*}/gi; let match; while ((match = fontFaceRegex.exec(cssBody)) !== null) { detectedFonts.add(match[1].trim().toLowerCase()); } } }); }); } function collectInternalLinks($) { const relativeLinks = $("a[href^='/'], a[href^='./'], a[href^='../']"); relativeLinks.each(function() { const absoluteLink = new URL($(this).attr('href'), baseUrl).href; if (!(absoluteLink in pagesVisited)) { pagesToVisit.push(absoluteLink); } }); }
Key Details About the Font Extraction Logic:
- We use a
Setto store fonts, which automatically handles duplicates and keeps entries unique. - Dual parsing: Cheerio for fast static CSS parsing, plus JSDOM to get computed styles (this accounts for CSS inheritance and priority, giving more accurate results).
- Covers inline styles, internal
<style>tags,@font-facedefinitions, and external stylesheets. - Added a custom
User-Agentheader to avoid being blocked by basic anti-scraping measures.
Getting this script running on a server is straightforward—here's how to do it:
Step 1: Prepare the Server
Ensure your server has Node.js and npm installed. Check with these commands:
node -v npm -v
If they're not installed, follow the official Node.js installation guide for your server's OS.
Step 2: Upload Your Script
Transfer the script file to your server using tools like scp, FTP, or Git. For example:
scp your-script-name.js user@your-server-ip:/path/to/directory/
Step 3: Install Dependencies
Navigate to the directory where your script is stored, then run:
npm install request cheerio url-parse jsdom
Step 4: Run the Script
- Basic run: For testing, run the script directly in your terminal:
node your-script-name.js - Background run (persistent): To keep the script running even after you close your SSH connection, use
pm2(a process manager):- Install pm2 globally:
npm install pm2 -g - Start the script:
pm2 start your-script-name.js --name font-crawler - Check logs or status:
pm2 logs font-crawler # View real-time logs pm2 status font-crawler # Check if the script is running - Stop the script when done:
pm2 stop font-crawler
- Install pm2 globally:
Quick Tips for Server Runs:
- Add a delay between requests to avoid overwhelming the target site (use
setTimeoutin the crawl function). - Adjust
MAX_PAGES_TO_VISITbased on the size of the site—100k is a large number, so start smaller if testing.
内容的提问来源于stack exchange,提问作者Alexander Solonik

