You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

为何我的谷歌图片搜索爬虫会超时?批量下载苹果图片遇阻

谷歌图片下载超时问题解决

尝试从谷歌图片搜索下载前10张苹果相关图片,打开搜索页面搜索“苹果”后,点击首张图片获取高清版本,但保存图片时频繁出现超时问题,原代码如下:

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
import time
import urllib


firefox_binary_path = 'Credentials/geckodriver.exe'
firefox_options = webdriver.FirefoxOptions()
firefox_options.binary_location = firefox_binary_path
driver = webdriver.Firefox()
driver.get('https://www.google.com/imghp')
search_box = driver.find_element("name", "q")
search_box.send_keys('Apples')
search_box.send_keys(Keys.RETURN)
downloaded_count = 0
try:
    while downloaded_count < 10:
        search_results = driver.find_elements(By.CSS_SELECTOR, '.rg_i')
        if downloaded_count < len(search_results):
            image = search_results[downloaded_count]
            ActionChains(driver).move_to_element(image).click(image).perform()
            img_locator = (By.CSS_SELECTOR, '.iPVvYb')
            WebDriverWait(driver, 10).until(EC.presence_of_element_located(img_locator))
            img = driver.find_element(*img_locator)
            img_url = img.get_attribute('src')
            resource = urllib.urlopen(img_url)
            filename = img_url.split('/')[-1] + '.jpg'
            output = open(filename, 'wb')
            output.write(resource.read())
            output.close()
            downloaded_count += 1
        else:
            driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
            time.sleep(2)
finally:
    driver.quit()

问题分析

  1. 请求无超时限制:urllib.urlopen(img_url)未设置超时,图片链接加载缓慢时会直接卡住
  2. 元素选择器失效:谷歌图片页面类名会定期更新,.iPVvYb已不是当前高清图的正确选择器
  3. 浏览器配置错误:firefox_options.binary_location应设置火狐浏览器的exe路径,而非geckodriver驱动路径
  4. 文件处理不严谨:直接用URL末尾作为文件名可能包含非法字符,且未处理读写异常

修改后的代码

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.common.action_chains import ActionChains
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.firefox.service import Service
import time
import requests
import os

# 配置火狐浏览器和驱动路径
firefox_binary_path = "C:/Program Files/Mozilla Firefox/firefox.exe"  # 替换为你的火狐安装路径
geckodriver_path = "Credentials/geckodriver.exe"  # 驱动文件路径

firefox_options = webdriver.FirefoxOptions()
firefox_options.binary_location = firefox_binary_path

# 初始化驱动(适配Selenium 4+)
service = Service(executable_path=geckodriver_path)
driver = webdriver.Firefox(service=service, options=firefox_options)
driver.get('https://www.google.com/imghp')

# 执行搜索
search_box = driver.find_element(By.NAME, "q")
search_box.send_keys('苹果')
search_box.send_keys(Keys.RETURN)

downloaded_count = 0
# 创建专属保存目录
save_dir = "apple_images"
os.makedirs(save_dir, exist_ok=True)

try:
    while downloaded_count < 10:
        # 获取当前加载的缩略图
        search_results = driver.find_elements(By.CSS_SELECTOR, '.rg_i.Q4LuWd')
        if downloaded_count >= len(search_results):
            # 滚动加载更多图片
            driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
            time.sleep(3)
            continue

        image = search_results[downloaded_count]
        try:
            # 点击缩略图打开大图
            ActionChains(driver).move_to_element(image).click(image).perform()
            
            # 等待大图完全可见(确保加载完成)
            img_locator = (By.CSS_SELECTOR, '.sFlh5c.pT0Scc.iPVvYb img')
            img = WebDriverWait(driver, 15).until(EC.visibility_of_element_located(img_locator))
            
            # 优先获取高清图链接(data-src),无则用src
            img_url = img.get_attribute('data-src') or img.get_attribute('src')
            if not img_url:
                print(f"第{downloaded_count+1}张图片无有效链接,跳过")
                downloaded_count += 1
                continue

            # 带超时请求图片
            response = requests.get(img_url, timeout=10)
            response.raise_for_status()  # 检查请求是否成功

            # 生成有序且安全的文件名
            filename = f"apple_{downloaded_count+1}.jpg"
            file_path = os.path.join(save_dir, filename)
            
            # 写入文件
            with open(file_path, 'wb') as f:
                f.write(response.content)
            
            print(f"已下载第{downloaded_count+1}张图片: {filename}")
            downloaded_count += 1

        except Exception as e:
            print(f"下载第{downloaded_count+1}张图片失败: {str(e)}")
            downloaded_count += 1
            continue

finally:
    driver.quit()
    print("下载完成,浏览器已关闭")

关键改进点

  • 替换请求库:用requests替代urllib,添加10秒超时限制,避免无限等待
  • 修正浏览器配置:区分火狐浏览器路径和驱动路径,适配Selenium 4+的Service初始化方式
  • 更新元素选择器:使用当前谷歌图片的大图选择器,优先获取data-src中的高清链接
  • 完善异常处理:捕获下载过程中的各类异常,跳过失败图片保证流程继续
  • 规范文件管理:创建专属保存目录,生成有序文件名,避免非法字符问题

内容的提问来源于stack exchange,提问作者A5omic

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.09 11:20:58