You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

基于Libcurl与C++的网页内容抓取失败问题排查求助

网页监控程序无法抓取内容的排查求助

我用Libcurl和C++开发了一款网页监控程序,需求是每秒抓取指定网页内容,存储后对比内容是否发生变化。目前程序看似已发起HTTP请求,但无法获取网页内容,控制台无内容输出,求帮忙排查原因。

#include <iostream>
#include <string>
#include <chrono>
#include <thread>
#include <curl/curl.h>
#include <Windows.h> // Include Windows API header for playing sound


using namespace std;



int sum=0;

//function to play sound
void playSound() {
    // Play sound using Windows API (Beep function)
    Beep(1000, 500); // Beep at 1000 Hz for 500 milliseconds
}


//the orifinal code before 
/*
// Function to perform HTTP request and fetch page content
static string fetchPageContent(const string& url) {
    CURL* curl;
    CURLcode res;
    string content;


    cout << "Making HTTP request to: " << url << endl;// making sure request is made

    // Initialize libcurl
    curl = curl_easy_init();
    if (curl) {
        // Set URL to fetch
        curl_easy_setopt(curl, CURLOPT_URL, url.c_str());
        // Set write callback function to store fetched content
        curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, [](void* buffer, size_t size, size_t nmemb, void* userp) -> size_t {
            ((string*)userp)->append((char*)buffer, size * nmemb);
            return size * nmemb;
            });
        // Set userp parameter to point to content string
        curl_easy_setopt(curl, CURLOPT_WRITEDATA, &content);
        // Perform HTTP request
        res = curl_easy_perform(curl);
        // Clean up
        curl_easy_cleanup(curl);
    }



    cout << "Fetched HTML content:" << endl;
    cout << content << endl; // Print fetched HTML content

    return content;
}
*/

//the new one to see why no content fetched 
// Function to perform HTTP request and fetch page content
static string fetchPageContent(const string& url) {
    CURL* curl;
    CURLcode res;
    string content;

    cout << "Making HTTP request to: " << url << endl;

    // Initialize libcurl
    curl = curl_easy_init();
    if (curl) {
        // Set URL to fetch
        curl_easy_setopt(curl, CURLOPT_URL, url.c_str());

        // Set SSL certificate verification
        curl_easy_setopt(curl, CURLOPT_SSL_VERIFYPEER, 1);
        curl_easy_setopt(curl, CURLOPT_SSL_VERIFYHOST, 2);

        // Set user agent string
        curl_easy_setopt(curl, CURLOPT_USERAGENT, "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/99 Safari/537.36");

        // Set timeout values
        curl_easy_setopt(curl, CURLOPT_TIMEOUT, 10); // 10 seconds timeout
        curl_easy_setopt(curl, CURLOPT_CONNECTTIMEOUT, 5); // 5 seconds connect timeout

        // Set write callback function to store fetched content
        curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, [](void* buffer, size_t size, size_t nmemb, void* userp) -> size_t {
            ((string*)userp)->append((char*)buffer, size * nmemb);
            return size * nmemb;
            });
        // Set userp parameter to point to content string
        curl_easy_setopt(curl, CURLOPT_WRITEDATA, &content);

        // Perform HTTP request
        res = curl_easy_perform(curl);

        // Check for errors
        if (res != CURLE_OK) {
            cerr << "Error during HTTP request: " << curl_easy_strerror(res) << endl;
        }

        // Clean up
        curl_easy_cleanup(curl);
    }

    cout << "Fetched HTML content:" << endl;
    cout << content << endl; // Print fetched HTML content

    return content;
}


// Function to parse HTML content and detect changes
bool parsePageContent(const string& previousContent, const string& currentContent) {
    // Compare previous and current HTML content
    return previousContent != currentContent;
}

/*the path adding of libcurl is done and now we have no errors in the basic layout code. 
Next step is to check if the data is getting parsed by adding the link. 
then we check if the logic is working( we have two logics) 
then I would have to create a same copy form web of my own. update it and see if the program is working 
or not. 
*/
int main() {

    string appointmentPageUrl = "https://docs.chocolatey.org/en-us/choco/setup#non-administrative-install";



    // Initialize previous HTML content
    string previousContent = "";

    // Main monitoring loop
    while (true) {
        // Fetch current page content
        string currentContent = fetchPageContent(appointmentPageUrl);

        // Parse page content and check for changes
        if (parsePageContent(previousContent, currentContent)) {
            // Display message in console if appointments are open
            cout << "open!" << endl;
            playSound();
            while (true)
            {  
                cout << "  open!  ";
                playSound();
            }
        }
        else
        {
            sum = sum + 1;
            cout << sum;
            system("cls");
            
        }
          //  system("cls");
            
        // Update previous HTML content
        previousContent = currentContent;

        // Sleep for a certain period before next check (e.g., 1 second)
        this_thread::sleep_for(chrono::seconds(1));
    }

    return 0;
}

可能的问题原因及解决方法

  • 缺少Libcurl全局初始化:Libcurl要求程序启动时必须调用curl_global_init(CURL_GLOBAL_ALL),否则HTTPS等核心功能会异常。在main函数开头添加:

    curl_global_init(CURL_GLOBAL_ALL);
    

    程序结束时可调用curl_global_cleanup();保证资源释放(规范操作)。

  • 控制台输出被快速清空:主循环中每次无变化时调用system("cls"),会直接清除之前的请求日志和内容输出,导致你看不到抓取结果。先注释掉system("cls")测试内容是否正常抓取,之后再调整输出逻辑(比如用光标移动刷新计数,而非清空整个控制台)。

  • HTTPS证书验证失败:如果系统未正确配置根证书,Libcurl的SSL验证会失败,导致无法获取内容。可临时关闭验证用于排查:

    curl_easy_setopt(curl, CURLOPT_SSL_VERIFYPEER, 0);
    curl_easy_setopt(curl, CURLOPT_SSL_VERIFYHOST, 0);
    

    确认问题后,再通过CURLOPT_CAINFO参数配置正确的根证书文件路径。

  • 输出缓冲未强制刷新:cout默认是行缓冲,可能内容还没输出就被清空。在cout << content << endl;后添加cout.flush();强制刷新缓冲区,确保内容能及时显示。

内容的提问来源于stack exchange,提问作者Sheikh Shessi

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 02:35:03