Chrome User Agent失效致图片爬虫无法抓取图片问题排查
Google图片爬取失败问题排查与修复
问题描述
我编写了以下爬取Google图片的代码,但运行后无法获取任何图片。我试过更换多个不同的User Agent(包括当前浏览器的真实UA),都没有效果。请问问题出在哪里?
import os, requests, lxml, re, json, urllib.request from bs4 import BeautifulSoup from os.path import expanduser headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/107.0.0.0 Safari/537.36" } params = { "q": "mincraft wallpaper 4k", # search query "tbm": "isch", # image results "hl": "en", # language of the search "gl": "us", # country where search comes from "ijn": "0" # page number } html = requests.get("https://www.google.com/search", params=params, headers=headers, timeout=30) soup = BeautifulSoup(html.text, "lxml") def get_original_images(): google_images = [] all_script_tags = soup.select("script") matched_images_data = "".join(re.findall(r"AF_initDataCallback\(([^<]+)\);", str(all_script_tags))) matched_images_data_fix = json.dumps(matched_images_data) matched_images_data_json = json.loads(matched_images_data_fix) matched_google_image_data = re.findall(r'\"b-GRID_STATE0\"(.*)sideChannel:\s?{}}', matched_images_data_json) matched_google_images_thumbnails = ", ".join( re.findall(r'\[\"(https\:\/\/encrypted-tbn0\.gstatic\.com\/images\?.*?)\",\d+,\d+\]', str(matched_google_image_data))).split(", ") thumbnails = [ bytes(bytes(thumbnail, "ascii").decode("unicode-escape"), "ascii").decode("unicode-escape") for thumbnail in matched_google_images_thumbnails ] # removing previously matched thumbnails for easier full resolution image matches. removed_matched_google_images_thumbnails = re.sub( r'\[\"(https\:\/\/encrypted-tbn0\.gstatic\.com\/images\?.*?)\",\d+,\d+\]', "", str(matched_google_image_data)) matched_google_full_resolution_images = re.findall(r"(?:'|,),\[\"(https:|http.*?)\",\d+,\d+\]", removed_matched_google_images_thumbnails) full_res_images = [ bytes(bytes(img, "ascii").decode("unicode-escape"), "ascii").decode("unicode-escape") for img in matched_google_full_resolution_images ] for index, (metadata, thumbnail, original) in enumerate( zip(soup.select('.isv-r.PNCib.MSM1fd.BUooTd'), thumbnails, full_res_images), start=1): google_images.append({ "title": metadata.select_one(".VFACy.kGQAp.sMi44c.lNHeqe.WGvvNb")["title"], "link": metadata.select_one(".VFACy.kGQAp.sMi44c.lNHeqe.WGvvNb")["href"], "source": metadata.select_one(".fxgdke").text, "thumbnail": thumbnail, "original": original }) # Download original images print(f'Downloading {index} image...') opener = urllib.request.build_opener() opener.addheaders = [('User-Agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/107.0.0.0 Safari/537.36')] urllib.request.install_opener(opener) urllib.request.urlretrieve(original, f'/Scrape/original_size_img_{index}.jpg') return google_images print(get_original_images())
问题根源
问题并非User Agent,而是以下几个核心问题:
- 正则匹配逻辑过时:Google图片页面的数据结构已更新,旧的
b-GRID_STATE0字段和正则规则无法匹配到有效数据,导致后续的图片链接全部为空。 - 页面元素选择器失效:代码中使用的元素class(如
.isv-r.PNCib.MSM1fd.BUooTd)已被Google修改,无法定位到图片元数据。 - 路径权限/存在性问题:
/Scrape/是系统根目录路径,多数情况下普通用户没有写入权限,且未提前创建该目录,会导致下载失败。
修复后的代码
import os import requests from bs4 import BeautifulSoup import json import re # 配置 headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36", "Accept-Language": "en-US,en;q=0.9" } params = { "q": "minecraft wallpaper 4k", # 修正拼写错误:mincraft → minecraft "tbm": "isch", "hl": "en", "gl": "us", "ijn": "0" } save_dir = "./Scrape" # 创建保存目录 os.makedirs(save_dir, exist_ok=True) def get_original_images(): google_images = [] html = requests.get("https://www.google.com/search", params=params, headers=headers, timeout=30) soup = BeautifulSoup(html.text, "lxml") # 提取图片数据的新逻辑 script_data = None for script in soup.select("script"): if "AF_initDataCallback" in script.text: # 提取并解析AF_initDataCallback中的数据 match = re.search(r"AF_initDataCallback\((.*?)\);", script.text, re.DOTALL) if match: raw_data = match.group(1).replace("'", '"') # 修复JSON格式问题 raw_data = re.sub(r",(\s*[\]}])", r"\1", raw_data) try: data = json.loads(raw_data) # 定位图片数据所在的层级 image_data = data[3][1][0][0][1][0] break except (json.JSONDecodeError, IndexError): continue if not image_data: print("未找到图片数据") return google_images # 遍历提取图片信息 for index, item in enumerate(image_data, start=1): try: original_url = item[0][3][0] thumbnail_url = item[0][2][0] title = item[0][1][0] source = item[0][1][2] link = item[0][1][1] google_images.append({ "title": title, "link": link, "source": source, "thumbnail": thumbnail_url, "original": original_url }) # 下载图片 print(f"Downloading {index} image...") img_response = requests.get(original_url, headers=headers, timeout=30) with open(os.path.join(save_dir, f"original_size_img_{index}.jpg"), "wb") as f: f.write(img_response.content) except (IndexError, requests.exceptions.RequestException) as e: print(f"第{index}张图片下载失败: {str(e)}") continue return google_images print(get_original_images())
修复说明
- 更新数据提取逻辑:适配Google当前的图片数据结构,从
AF_initDataCallback中正确解析出图片链接和元数据。 - 修复路径问题:使用相对路径
./Scrape,并自动创建目录,避免权限和不存在的问题。 - 更新元素解析方式:不再依赖页面DOM元素,直接从脚本数据中提取信息,更稳定。
- 修正拼写错误:搜索关键词
mincraft改为minecraft,避免无效搜索结果。 - 增加异常处理:捕获下载和解析过程中的错误,避免程序崩溃。
内容的提问来源于stack exchange,提问作者v_head
相关产品推荐
相关产品推荐

