为何我的谷歌图片搜索爬虫会超时?批量下载苹果图片遇阻
谷歌图片下载超时问题解决
尝试从谷歌图片搜索下载前10张苹果相关图片,打开搜索页面搜索“苹果”后,点击首张图片获取高清版本,但保存图片时频繁出现超时问题,原代码如下:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.common.action_chains import ActionChains from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.wait import WebDriverWait import time import urllib firefox_binary_path = 'Credentials/geckodriver.exe' firefox_options = webdriver.FirefoxOptions() firefox_options.binary_location = firefox_binary_path driver = webdriver.Firefox() driver.get('https://www.google.com/imghp') search_box = driver.find_element("name", "q") search_box.send_keys('Apples') search_box.send_keys(Keys.RETURN) downloaded_count = 0 try: while downloaded_count < 10: search_results = driver.find_elements(By.CSS_SELECTOR, '.rg_i') if downloaded_count < len(search_results): image = search_results[downloaded_count] ActionChains(driver).move_to_element(image).click(image).perform() img_locator = (By.CSS_SELECTOR, '.iPVvYb') WebDriverWait(driver, 10).until(EC.presence_of_element_located(img_locator)) img = driver.find_element(*img_locator) img_url = img.get_attribute('src') resource = urllib.urlopen(img_url) filename = img_url.split('/')[-1] + '.jpg' output = open(filename, 'wb') output.write(resource.read()) output.close() downloaded_count += 1 else: driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2) finally: driver.quit()
问题分析
- 请求无超时限制:
urllib.urlopen(img_url)未设置超时,图片链接加载缓慢时会直接卡住 - 元素选择器失效:谷歌图片页面类名会定期更新,
.iPVvYb已不是当前高清图的正确选择器 - 浏览器配置错误:
firefox_options.binary_location应设置火狐浏览器的exe路径,而非geckodriver驱动路径 - 文件处理不严谨:直接用URL末尾作为文件名可能包含非法字符,且未处理读写异常
修改后的代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.common.action_chains import ActionChains from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.support.wait import WebDriverWait from selenium.webdriver.firefox.service import Service import time import requests import os # 配置火狐浏览器和驱动路径 firefox_binary_path = "C:/Program Files/Mozilla Firefox/firefox.exe" # 替换为你的火狐安装路径 geckodriver_path = "Credentials/geckodriver.exe" # 驱动文件路径 firefox_options = webdriver.FirefoxOptions() firefox_options.binary_location = firefox_binary_path # 初始化驱动(适配Selenium 4+) service = Service(executable_path=geckodriver_path) driver = webdriver.Firefox(service=service, options=firefox_options) driver.get('https://www.google.com/imghp') # 执行搜索 search_box = driver.find_element(By.NAME, "q") search_box.send_keys('苹果') search_box.send_keys(Keys.RETURN) downloaded_count = 0 # 创建专属保存目录 save_dir = "apple_images" os.makedirs(save_dir, exist_ok=True) try: while downloaded_count < 10: # 获取当前加载的缩略图 search_results = driver.find_elements(By.CSS_SELECTOR, '.rg_i.Q4LuWd') if downloaded_count >= len(search_results): # 滚动加载更多图片 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(3) continue image = search_results[downloaded_count] try: # 点击缩略图打开大图 ActionChains(driver).move_to_element(image).click(image).perform() # 等待大图完全可见(确保加载完成) img_locator = (By.CSS_SELECTOR, '.sFlh5c.pT0Scc.iPVvYb img') img = WebDriverWait(driver, 15).until(EC.visibility_of_element_located(img_locator)) # 优先获取高清图链接(data-src),无则用src img_url = img.get_attribute('data-src') or img.get_attribute('src') if not img_url: print(f"第{downloaded_count+1}张图片无有效链接,跳过") downloaded_count += 1 continue # 带超时请求图片 response = requests.get(img_url, timeout=10) response.raise_for_status() # 检查请求是否成功 # 生成有序且安全的文件名 filename = f"apple_{downloaded_count+1}.jpg" file_path = os.path.join(save_dir, filename) # 写入文件 with open(file_path, 'wb') as f: f.write(response.content) print(f"已下载第{downloaded_count+1}张图片: {filename}") downloaded_count += 1 except Exception as e: print(f"下载第{downloaded_count+1}张图片失败: {str(e)}") downloaded_count += 1 continue finally: driver.quit() print("下载完成,浏览器已关闭")
关键改进点
- 替换请求库:用
requests替代urllib,添加10秒超时限制,避免无限等待 - 修正浏览器配置:区分火狐浏览器路径和驱动路径,适配Selenium 4+的Service初始化方式
- 更新元素选择器:使用当前谷歌图片的大图选择器,优先获取
data-src中的高清链接 - 完善异常处理:捕获下载过程中的各类异常,跳过失败图片保证流程继续
- 规范文件管理:创建专属保存目录,生成有序文件名,避免非法字符问题
内容的提问来源于stack exchange,提问作者A5omic
相关产品推荐
相关产品推荐

