You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

使用代理仍遇Error 429请求过多?求解决方案

解决Google搜索频繁触发429错误的问题

我编写了一段用于搜索企业Facebook商业主页的Python代码,已经配置了每次搜索轮换User-Agent和代理,但仍频繁触发Error 429请求过多错误。代码功能正常,仅该错误阻碍运行,请问如何解决?是否是代理的问题?

import openpyxl
import time
import requests
from bs4 import BeautifulSoup
from fake_useragent import UserAgent

def search_google(query):
    search_query = f"{query} facebook business profile"
    google_url = f"https://www.google.com/search?q={search_query}"

    ua = UserAgent()
    headers = {
        "User-Agent": ua.random
    }

    proxy = {
        "http": "http://elyaiwoq-rotate:flw5qf82idr6@p.webshare.io:80/",
        "https": "http://elyaiwoq-rotate:flw5qf82idr6@p.webshare.io:80/"
    }

    try:
        response = requests.get(google_url, headers=headers, proxies=proxy)
        time.sleep(10)  # Add a delay of 10 seconds between each search request

        if response.status_code == 200:
            return response.content
        else:
            raise Exception(f"Failed to retrieve Google search results. Status code: {response.status_code}")

    except Exception as e:
        raise Exception(f"An error occurred with the proxy {proxy['http']}: {e}")

def extract_facebook_profiles(html_content):
    soup = BeautifulSoup(html_content, "html.parser")

    profile_links = []

    for link in soup.find_all("a"):
        href = link.get("href")
        if href and "facebook.com/" in href:
            profile_links.append(href)

    return profile_links

def read_excel_file(file_name):
    wb = openpyxl.load_workbook(file_name)
    sheet = wb.active
    business_names = [cell.value for row in sheet.iter_rows() for cell in row]
    return business_names

def main():
    try:
        input_file_name = "business_names.xlsx"  # Replace with the name of your input Excel file
        output_file_name = "facebook_profiles.xlsx"  # Replace with the name of your output Excel file

        business_names = read_excel_file(input_file_name)
        all_results = []

        for business_name in business_names:
            print(f"Searching for Facebook business profile of '{business_name}' using the provided proxy...")
            try:
                search_results = search_google(business_name)
                profile_links = extract_facebook_profiles(search_results)

                if profile_links:
                    all_results.append([business_name, profile_links[0]])
                    print(f"Facebook profile found: {profile_links[0]}")
                else:
                    all_results.append([business_name, "Not found"])
                    print("Facebook profile not found.")
            except Exception as e:
                print(f"Error occurred with the proxy: {e}")
            print("-" * 50)

        if all_results:
            wb = openpyxl.Workbook()
            sheet = wb.active
            sheet.append(["Business Name", "Facebook Profile"])

            for result in all_results:
                sheet.append(result)

            wb.save(output_file_name)
            print(f"Search results saved to '{output_file_name}'.")

    except Exception as e:
        print(f"An error occurred: {e}")

if __name__ == "__main__":
    main()

问题分析与解决办法

1. 代理大概率是核心问题

你当前用的是共享轮换代理,这类代理池里的IP通常被大量用户同时使用,很容易被Google的反爬系统标记为异常IP,直接触发429限制。

  • 解决措施:
    • 更换为私有代理池,避免使用公开共享代理;
    • 每次请求前先验证代理有效性,比如先请求Google首页,确认能正常访问再用;
    • 不要依赖服务商的自动轮换,自己维护代理列表,每次请求主动切换不同IP。

2. 请求延迟设置不合理

你把time.sleep(10)放在请求之后,而且是固定延迟,这种机械的间隔很容易被识别为爬虫。

  • 解决措施:
    • 改用随机延迟,比如time.sleep(random.randint(15, 35)),模拟人类不规则的操作间隔;
    • 把延迟移到两次搜索之间(也就是main函数的循环里),而不是请求完成后;
    • 批量搜索后增加长休息,比如每搜10个企业,休息1-2分钟。

3. 请求头太简陋

只设置User-Agent远远不够,Google会检查多个请求头字段,缺少关键字段会被判定为非人类请求。

  • 解决措施:
    补充完整的请求头,示例如下:
    headers = {
        "User-Agent": ua.random,
        "Accept-Language": "en-US,en;q=0.9",
        "Accept-Encoding": "gzip, deflate, br",
        "Referer": "https://www.google.com/",
        "DNT": "1",
        "Connection": "keep-alive"
    }
    

4. 缺少错误重试机制

当前代码碰到429直接抛出错误,没有重试逻辑。碰到429时,应该换代理、换UA后再尝试请求。

  • 解决措施:
    • 给请求添加重试逻辑,比如用tenacity库实现,指定仅在碰到429或代理错误时重试;
    • 简单版可以手动捕获429状态码,切换代理后等待更长时间再重试。

5. 搜索行为太机械

连续搜索相似关键词、直接拼接简单URL,很容易被Google识别。

  • 解决措施:
    • 在搜索URL中添加真实参数,比如hl=en-US(语言)、gl=us(地区),模拟本地化搜索;
    • 偶尔打乱搜索顺序,或者插入几个无关搜索,避免连续相同行为。

优化后的代码示例(核心改动)

import openpyxl
import time
import random
import requests
from bs4 import BeautifulSoup
from fake_useragent import UserAgent

# 维护私有代理列表
PROXY_LIST = [
    "http://proxy1:port",
    "http://proxy2:port",
    # 添加更多代理...
]

def get_random_proxy():
    return random.choice(PROXY_LIST)

def search_google(query):
    search_query = f"{query} facebook business profile"
    # 增加真实搜索参数
    google_url = f"https://www.google.com/search?q={search_query}&hl=en-US&gl=us"

    ua = UserAgent()
    # 完整请求头
    headers = {
        "User-Agent": ua.random,
        "Accept-Language": "en-US,en;q=0.9",
        "Accept-Encoding": "gzip, deflate, br",
        "Referer": "https://www.google.com/",
        "DNT": "1",
        "Connection": "keep-alive"
    }

    proxy = {
        "http": get_random_proxy(),
        "https": get_random_proxy()
    }

    try:
        response = requests.get(google_url, headers=headers, proxies=proxy, timeout=15)
        
        if response.status_code == 200:
            return response.content
        elif response.status_code == 429:
            raise Exception(f"429 Too Many Requests. Proxy {proxy['http']} may be blocked.")
        else:
            raise Exception(f"Failed. Status code: {response.status_code}")

    except Exception as e:
        raise Exception(f"Proxy error: {e}")

def extract_facebook_profiles(html_content):
    soup = BeautifulSoup(html_content, "html.parser")
    profile_links = []

    for link in soup.find_all("a"):
        href = link.get("href")
        if href and "facebook.com/" in href:
            # 解析Google跳转的真实链接
            if href.startswith("/url?q="):
                href = href.split("/url?q=")[1].split("&")[0]
            profile_links.append(href)

    return profile_links

def read_excel_file(file_name):
    wb = openpyxl.load_workbook(file_name)
    sheet = wb.active
    # 过滤空值
    business_names = [cell.value for row in sheet.iter_rows() for cell in row if cell.value]
    return business_names

def main():
    try:
        input_file_name = "business_names.xlsx"
        output_file_name = "facebook_profiles.xlsx"

        business_names = read_excel_file(input_file_name)
        all_results = []

        for idx, business_name in enumerate(business_names):
            print(f"[{idx+1}/{len(business_names)}] Searching for '{business_name}'...")
            retries = 3
            success = False
            while retries > 0 and not success:
                try:
                    search_results = search_google(business_name)
                    profile_links = extract_facebook_profiles(search_results)

                    if profile_links:
                        all_results.append([business_name, profile_links[0]])
                        print(f"Found: {profile_links[0]}")
                    else:
                        all_results.append([business_name, "Not found"])
                        print("Not found.")
                    success = True
                except Exception as e:
                    retries -= 1
                    print(f"Error (retries left: {retries}): {e}")
                    time.sleep(random.randint(20, 40))
            
            # 两次搜索间的随机延迟
            time.sleep(random.randint(15, 30))
            # 每10次搜索后长休息
            if (idx+1) % 10 == 0:
                print("Taking a long break...")
                time.sleep(random.randint(60, 120))
            print("-" * 50)

        if all_results:
            wb = openpyxl.Workbook()
            sheet = wb.active
            sheet.append(["Business Name", "Facebook Profile"])

            for result in all_results:
                sheet.append(result)

            wb.save(output_file_name)
            print(f"Results saved to '{output_file_name}'.")

    except Exception as e:
        print(f"Fatal error: {e}")

if __name__ == "__main__":
    main()

内容的提问来源于stack exchange,提问作者Ian Terry

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.14 23:00:55