You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在C语言中无需API获取Google‘Did you mean’数据并保存搜索页源码

Hey there! Let's tackle your two C language needs one by one—I’ve helped plenty of developers with similar web scraping and parsing tasks before.

1. Saving Google Search Page Source with C

It sounds like your curl attempt failed because Google blocks requests that don’t identify themselves as a real browser. Let’s fix that with a proper curl implementation that includes necessary headers and handles basic anti-scraping checks.

Here’s a complete working example that saves the search page source to a file:

#include <stdio.h>
#include <curl/curl.h>

// Callback function to write fetched data to a file
size_t write_to_file(void *ptr, size_t size, size_t nmemb, FILE *stream) {
    return fwrite(ptr, size, nmemb, stream);
}

int main(void) {
    CURL *curl_handle;
    CURLcode request_status;
    FILE *output_file;
    const char *target_url = "https://www.google.com/search?q=stacoverflow";
    // Use a real browser's User-Agent to avoid being blocked
    const char *user_agent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/118.0.0.0 Safari/537.36";

    // Initialize curl
    curl_handle = curl_easy_init();
    if (!curl_handle) {
        fprintf(stderr, "Failed to initialize curl\n");
        return 1;
    }

    // Open output file for writing
    output_file = fopen("google_search_source.html", "wb");
    if (!output_file) {
        fprintf(stderr, "Failed to open output file\n");
        curl_easy_cleanup(curl_handle);
        return 1;
    }

    // Configure curl options
    curl_easy_setopt(curl_handle, CURLOPT_URL, target_url);
    curl_easy_setopt(curl_handle, CURLOPT_USERAGENT, user_agent);
    curl_easy_setopt(curl_handle, CURLOPT_WRITEFUNCTION, write_to_file);
    curl_easy_setopt(curl_handle, CURLOPT_WRITEDATA, output_file);
    // Bypass SSL checks temporarily (fix certificate issues in production!)
    curl_easy_setopt(curl_handle, CURLOPT_SSL_VERIFYPEER, 0L);
    curl_easy_setopt(curl_handle, CURLOPT_SSL_VERIFYHOST, 0L);

    // Execute the request
    request_status = curl_easy_perform(curl_handle);
    if (request_status != CURLE_OK) {
        fprintf(stderr, "Request failed: %s\n", curl_easy_strerror(request_status));
    }

    // Clean up resources
    fclose(output_file);
    curl_easy_cleanup(curl_handle);
    return 0;
}

To compile and run this:

gcc -o save_google_source save_google_source.c -lcurl
./save_google_source

You’ll find the page source in google_search_source.html, which matches what you see in view-source:https://www.google.com/search?q=stacoverflow.

2. Extracting "Did you mean" / "Showing results for" Without APIs

Once you have the page source, you need to parse it to extract the spelling suggestion. Google’s HTML structure changes often, but here’s a reliable approach using string manipulation (for quick use cases) or a proper HTML parser for long-term stability.

Simple String Matching Approach

This uses strstr to locate key parts of the HTML and extract the suggestion:

#include <stdio.h>
#include <string.h>
#include <stdlib.h>

// Simple URL decoder to fix encoded spaces/characters
void url_decode(char *str) {
    char *dst = str;
    while (*str) {
        if (*str == '%' && *(str+1) && *(str+2)) {
            // Convert %XX hex values to ASCII
            int hex_val = 0;
            hex_val += (*(str+1) >= 'A' ? (*(str+1)-'A'+10) : (*(str+1)-'0')) * 16;
            hex_val += (*(str+2) >= 'A' ? (*(str+2)-'A'+10) : (*(str+2)-'0'));
            *dst++ = hex_val;
            str += 3;
        } else if (*str == '+') {
            *dst++ = ' ';
            str++;
        } else {
            *dst++ = *str++;
        }
    }
    *dst = '\0';
}

void extract_spelling_suggestion(const char *html) {
    // Check for both possible suggestion phrases
    const char *suggestion_start = strstr(html, "Did you mean");
    if (!suggestion_start) {
        suggestion_start = strstr(html, "Showing results for");
        if (!suggestion_start) {
            printf("No spelling suggestion found.\n");
            return;
        }
    }

    // Locate the link containing the corrected query
    const char *link_start = strstr(suggestion_start, "<a href=\"/search?q=");
    if (!link_start) {
        printf("Could not find suggestion link.\n");
        return;
    }

    // Skip to the start of the query parameter
    link_start += strlen("<a href=\"/search?q=");
    const char *link_end = strstr(link_start, "\"");
    if (!link_end) {
        printf("Could not parse suggestion link.\n");
        return;
    }

    // Extract and decode the suggestion
    char suggestion[256];
    strncpy(suggestion, link_start, link_end - link_start);
    suggestion[link_end - link_start] = '\0';
    url_decode(suggestion);
    printf("Spelling Suggestion: %s\n", suggestion);
}

int main(void) {
    // Read the saved HTML file
    FILE *html_file = fopen("google_search_source.html", "r");
    if (!html_file) {
        fprintf(stderr, "Failed to open HTML file.\n");
        return 1;
    }

    // Get file size to allocate memory
    fseek(html_file, 0, SEEK_END);
    long file_size = ftell(html_file);
    fseek(html_file, 0, SEEK_SET);

    char *html_content = malloc(file_size + 1);
    if (!html_content) {
        fprintf(stderr, "Failed to allocate memory.\n");
        fclose(html_file);
        return 1;
    }

    fread(html_content, 1, file_size, html_file);
    html_content[file_size] = '\0';

    // Extract the suggestion
    extract_spelling_suggestion(html_content);

    // Clean up
    free(html_content);
    fclose(html_file);
    return 0;
}

Notes for Reliability

  • Google’s HTML structure can change without warning, so this string-based method might break over time. For a more robust solution, use an HTML parser like libxml2 with XPath queries to target elements by class or ID (you’ll need to inspect Google’s current HTML to find stable selectors).
  • Always comply with Google’s Terms of Service and robots.txt—don’t send too many requests in a short time, or your IP may get blocked.

内容的提问来源于stack exchange,提问作者ahmetelgun

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.05.26 08:58:24