You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Node.js环境下获取网页用户输入并用于网页爬虫动态搜索

解决方案

你现有代码存在两个核心问题:

  • 关键词SEARCH_TERM写死为固定值,且爬虫逻辑在服务启动时就一次性执行,无法根据用户输入动态调整
  • 全局存储articles数组,所有用户访问拿到的都是服务启动时爬取的固定结果

具体实现

1. 完整修正代码

"use strict"
const PORT = process.env.PORT || 8000
const express = require('express')
const axios = require('axios')
const cheerio = require('cheerio')

const app = express()
app.use(express.urlencoded({ extended: true }))

const news = [
    {
        name: 'google news',
        address: 'https://news.google.com/topstories?hl=en-US&gl=US&ceid=US:en',
        base:'https://news.google.com'
    },
    {
        name: 'yahoo news',
        address: 'https://news.yahoo.com/',
        base: 'https://news.yahoo.com'
    },
    {
        name: 'huffington post',
        address: 'https://www.huffpost.com/',
        base: 'https://www.huffpost.com'
    },
    {
        name: 'cnn',
        address: 'https://www.cnn.com/',
        base: 'https://www.cnn.com'
    },
    {
        name: 'the guardian',
        address: 'https://www.theguardian.com/us',
        base: 'https://www.theguardian.com'
    },
    {
        name: 'usa today',
        address: 'https://www.usatoday.com/',
        base: 'https://www.usatoday.com'
    },
    {
        name: 'business insider',
        address: 'https://www.businessinsider.com/',
        base: 'https://www.businessinsider.com'
    },
    {
        name: 'bbc',
        address: 'https://www.bbc.com/',
        base: 'https://www.bbc.com'
    },
    {
        name: 'vice',
        address: 'https://www.vice.com/',
        base: 'https://www.vice.com'
    },
    {
        name: 'the new york post',
        address: 'https://nypost.com/',
        base: 'https://nypost.com'
    },
    {
        name: 'vox',
        address: 'https://www.vox.com/',
        base: 'https://www.vox.com'
    },
    {
        name: 'the atlantic',
        address: 'https://www.theatlantic.com/',
        base: 'https://www.theatlantic.com'
    },
    {
        name: 'the times of india',
        address: 'https://timesofindia.indiatimes.com/us',
        base: 'https://timesofindia.indiatimes.com'
    },
    {
        name: 'china daily',
        address: 'http://global.chinadaily.com.cn/',
        base: 'http://global.chinadaily.com.cn'
    },
    {
        name: 'the hindu',
        address: 'https://www.thehindu.com/',
        base: 'https://www.thehindu.com'
    },
    {
        name: 'the south china morning post',
        address: 'https://www.scmp.com/',
        base: 'https://www.scmp.com'
    },
    {
        name: 'al jazeera',
        address: 'https://www.aljazeera.com',
        base: 'https://www.aljazeera.com'
    },
]

// 封装爬虫逻辑,接收关键词返回结果
async function getNewsByKeyword(keyword) {
    const articles = []
    const requests = news.map(async newspaper => {
        try {
            const response = await axios.get(newspaper.address, { timeout: 5000 })
            const html = response.data
            const $ = cheerio.load(html)
            // 不区分大小写匹配关键词,避免漏抓
            $(`a`, html).each(function() {
                const title = $(this).text().trim()
                if (title.toLowerCase().includes(keyword.toLowerCase())) {
                    let url = $(this).attr('href')
                    if (url?.startsWith("/") || url?.startsWith(".")) {
                        url = newspaper.base + url
                    }
                    if (!url || url.startsWith("/")) {
                        url = "NO_URL"
                    }
                    url = url.replace(/\s+/g, '')
                    articles.push({
                        title,
                        url: url,
                        source_website: newspaper.name
                    })
                }
            })
        } catch (e) {
            console.log(`爬取${newspaper.name}失败:`, e.message)
        }
    })
    // 等待所有网站爬取完成
    await Promise.all(requests)
    return articles
}

// 根路由返回带输入提示的页面
app.get('/', (req, res) => {
    res.send(`
    <script>
        window.onload = function() {
            const keyword = prompt('Enter your search term: ')
            if (keyword) {
                window.location.href = '/news?keyword=' + encodeURIComponent(keyword)
            }
        }
    </script>
    `)
})

// 新闻查询接口
app.get('/news', async (req, res) => {
    const { keyword } = req.query
    if (!keyword) {
        return res.status(400).json({ msg: '请传入搜索关键词' })
    }
    const articles = await getNewsByKeyword(keyword)
    res.json(articles)
})

app.listen(PORT, () => console.log('server running on PORT ' + PORT))

2. 改动说明

  • 移除了全局固定的SEARCH_TERM和预执行的爬虫逻辑,把爬虫封装为getNewsByKeyword异步函数,接收关键词作为入参
  • 新增根路由/,返回的页面加载时会自动弹出输入提示框,要求用户输入搜索关键词
  • 调整/news接口为动态接收keyword查询参数,调用爬取函数后返回对应结果
  • 新增了请求超时和错误捕获逻辑,避免单个新闻网站请求失败导致整个服务报错
  • 优化了关键词匹配逻辑为不区分大小写,避免漏抓标题大小写不一致的内容

3. 运行测试

  • 安装依赖:npm install express axios cheerio
  • 启动服务:node 你的脚本文件名.js
  • 浏览器访问http://localhost:8000,页面会自动弹出输入提示,输入关键词后等待几秒即可返回对应爬取结果

内容的提问来源于stack exchange,提问作者Stan Preschlack

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.09.27 06:36:04