You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Chrome User Agent失效致图片爬虫无法抓取图片问题排查

Google图片爬取失败问题排查与修复

问题描述

我编写了以下爬取Google图片的代码,但运行后无法获取任何图片。我试过更换多个不同的User Agent(包括当前浏览器的真实UA),都没有效果。请问问题出在哪里?

import os, requests, lxml, re, json, urllib.request
from bs4 import BeautifulSoup
from os.path import expanduser

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/107.0.0.0 Safari/537.36"
}

params = {
    "q": "mincraft wallpaper 4k",  # search query
    "tbm": "isch",  # image results
    "hl": "en",  # language of the search
    "gl": "us",  # country where search comes from
    "ijn": "0"  # page number
}

html = requests.get("https://www.google.com/search", params=params, headers=headers, timeout=30)
soup = BeautifulSoup(html.text, "lxml")

def get_original_images():
    google_images = []
    all_script_tags = soup.select("script")
    matched_images_data = "".join(re.findall(r"AF_initDataCallback\(([^<]+)\);", str(all_script_tags)))
    matched_images_data_fix = json.dumps(matched_images_data)
    matched_images_data_json = json.loads(matched_images_data_fix)
    matched_google_image_data = re.findall(r'\&quot;b-GRID_STATE0\&quot;(.*)sideChannel:\s?{}}', matched_images_data_json)
    matched_google_images_thumbnails = ", ".join(
        re.findall(r'\[\&quot;(https\:\/\/encrypted-tbn0\.gstatic\.com\/images\?.*?)\&quot;,\d+,\d+\]',
                   str(matched_google_image_data))).split(", ")
    thumbnails = [
        bytes(bytes(thumbnail, "ascii").decode("unicode-escape"), "ascii").decode("unicode-escape") for thumbnail in
        matched_google_images_thumbnails
    ]
    # removing previously matched thumbnails for easier full resolution image matches.
    removed_matched_google_images_thumbnails = re.sub(
        r'\[\&quot;(https\:\/\/encrypted-tbn0\.gstatic\.com\/images\?.*?)\&quot;,\d+,\d+\]', "", str(matched_google_image_data))

    matched_google_full_resolution_images = re.findall(r"(?:'|,),\[\&quot;(https:|http.*?)\&quot;,\d+,\d+\]",
                                                       removed_matched_google_images_thumbnails)
    full_res_images = [
        bytes(bytes(img, "ascii").decode("unicode-escape"), "ascii").decode("unicode-escape") for img in
        matched_google_full_resolution_images
    ]
    for index, (metadata, thumbnail, original) in enumerate(
            zip(soup.select('.isv-r.PNCib.MSM1fd.BUooTd'), thumbnails, full_res_images), start=1):
        google_images.append({
            "title": metadata.select_one(".VFACy.kGQAp.sMi44c.lNHeqe.WGvvNb")["title"],
            "link": metadata.select_one(".VFACy.kGQAp.sMi44c.lNHeqe.WGvvNb")["href"],
            "source": metadata.select_one(".fxgdke").text,
            "thumbnail": thumbnail,
            "original": original
        })
        # Download original images
        print(f'Downloading {index} image...')
        opener = urllib.request.build_opener()
        opener.addheaders = [('User-Agent',
                              'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/107.0.0.0 Safari/537.36')]
        urllib.request.install_opener(opener)

        urllib.request.urlretrieve(original, f'/Scrape/original_size_img_{index}.jpg')

    return google_images
print(get_original_images())

问题根源

问题并非User Agent,而是以下几个核心问题:

  • 正则匹配逻辑过时:Google图片页面的数据结构已更新,旧的b-GRID_STATE0字段和正则规则无法匹配到有效数据,导致后续的图片链接全部为空。
  • 页面元素选择器失效:代码中使用的元素class(如.isv-r.PNCib.MSM1fd.BUooTd)已被Google修改,无法定位到图片元数据。
  • 路径权限/存在性问题:/Scrape/是系统根目录路径,多数情况下普通用户没有写入权限,且未提前创建该目录,会导致下载失败。

修复后的代码

import os
import requests
from bs4 import BeautifulSoup
import json
import re

# 配置
headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36",
    "Accept-Language": "en-US,en;q=0.9"
}
params = {
    "q": "minecraft wallpaper 4k",  # 修正拼写错误:mincraft → minecraft
    "tbm": "isch",
    "hl": "en",
    "gl": "us",
    "ijn": "0"
}
save_dir = "./Scrape"

# 创建保存目录
os.makedirs(save_dir, exist_ok=True)

def get_original_images():
    google_images = []
    html = requests.get("https://www.google.com/search", params=params, headers=headers, timeout=30)
    soup = BeautifulSoup(html.text, "lxml")

    # 提取图片数据的新逻辑
    script_data = None
    for script in soup.select("script"):
        if "AF_initDataCallback" in script.text:
            # 提取并解析AF_initDataCallback中的数据
            match = re.search(r"AF_initDataCallback\((.*?)\);", script.text, re.DOTALL)
            if match:
                raw_data = match.group(1).replace("'", '"')
                # 修复JSON格式问题
                raw_data = re.sub(r",(\s*[\]}])", r"\1", raw_data)
                try:
                    data = json.loads(raw_data)
                    # 定位图片数据所在的层级
                    image_data = data[3][1][0][0][1][0]
                    break
                except (json.JSONDecodeError, IndexError):
                    continue

    if not image_data:
        print("未找到图片数据")
        return google_images

    # 遍历提取图片信息
    for index, item in enumerate(image_data, start=1):
        try:
            original_url = item[0][3][0]
            thumbnail_url = item[0][2][0]
            title = item[0][1][0]
            source = item[0][1][2]
            link = item[0][1][1]

            google_images.append({
                "title": title,
                "link": link,
                "source": source,
                "thumbnail": thumbnail_url,
                "original": original_url
            })

            # 下载图片
            print(f"Downloading {index} image...")
            img_response = requests.get(original_url, headers=headers, timeout=30)
            with open(os.path.join(save_dir, f"original_size_img_{index}.jpg"), "wb") as f:
                f.write(img_response.content)

        except (IndexError, requests.exceptions.RequestException) as e:
            print(f"第{index}张图片下载失败: {str(e)}")
            continue

    return google_images

print(get_original_images())

修复说明

  1. 更新数据提取逻辑:适配Google当前的图片数据结构,从AF_initDataCallback中正确解析出图片链接和元数据。
  2. 修复路径问题:使用相对路径./Scrape,并自动创建目录,避免权限和不存在的问题。
  3. 更新元素解析方式:不再依赖页面DOM元素,直接从脚本数据中提取信息,更稳定。
  4. 修正拼写错误:搜索关键词mincraft改为minecraft,避免无效搜索结果。
  5. 增加异常处理:捕获下载和解析过程中的错误,避免程序崩溃。

内容的提问来源于stack exchange,提问作者v_head

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.11 04:25:32