使用Python+Selenium提取亚马逊商品图片URL报错排查
问题描述
我想要从亚马逊印度站提取商品的所有关联图片(假设有8个媒体文件),获取其img标签的src属性。使用Python结合Selenium实现,思路是定位id为main-image-container的商品媒体主容器,提取其中class为imgTagWrapper内的图片URL,商品标题来自CSV文件。
以下是我当前使用的代码:
import pandas as pd from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time def setup_driver(): """Setup Chrome WebDriver with options.""" options = Options() options.headless = True # Optional: Run in headless mode if no browser UI needed service = Service(executable_path=r'C:\Users\preri\Desktop\Yorkee\chromedriver.exe') return webdriver.Chrome(service=service, options=options) def fetch_images(title, driver): """Fetch image URLs for a given product title from Amazon within 'imgTagWrapper'.""" try: driver.get("https://www.amazon.in") search_box = driver.find_element(By.NAME, "field-keywords") search_box.send_keys(title + Keys.RETURN) time.sleep(5) # Allow time for search results to load # Wait for product links to be visible and interactable product_links = WebDriverWait(driver, 10).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, 'a.a-link-normal.s-link-style')) ) if product_links: product_links[0].click() # Click the first product link if available WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.ID, "main-image-container")) ) # Specific fetch within 'imgTagWrapper' images = driver.find_elements(By.CSS_SELECTOR, '#main-image-container .imgTagWrapper img') return [img.get_attribute('src') for img in images if img.get_attribute('src')] else: print(f"No product links found for {title}") return [] except Exception as e: print(f"Error fetching images for {title}: {e}") return [] finally: driver.quit() def main(): driver = setup_driver() df = pd.read_csv(r'C:\Users\preri\Desktop\Yorkee\Yorkee Sample Prod.csv') df['Title'] = df['Title'].astype(str) # Convert all titles to string df['Image URLs'] = df['Title'].apply(lambda title: fetch_images(title, driver)) df.to_csv('modified_file_with_images.csv', index=False) print("All entries processed and saved.") if __name__ == "__main__": main()
运行代码时出现如下错误:
Error fetching images for Anandlib Na Sahi Sahil To Hai: Message: Stacktrace: GetHandleVerifier [0x00007FF755EA1502+60802] (No symbol) [0x00007FF755E1AC02] (No symbol) [0x00007FF755CD7CE4] (No symbol) [0x00007FF755D26D4D] (No symbol) [0x00007FF755D26E1C] (No symbol) [0x00007FF755D6CE37] (No symbol) [0x00007FF755D4ABBF] (No symbol) [0x00007FF755D6A224] (No symbol) [0x00007FF755D4A923] (No symbol) [0x00007FF755D18FEC] (No symbol) [0x00007FF755D19C21] GetHandleVerifier [0x00007FF7561A411D+3217821] GetHandleVerifier [0x00007FF7561E60B7+3488055] GetHandleVerifier [0x00007FF7561DF03F+3459263] GetHandleVerifier [0x00007FF755F5B846+823494] (No symbol) [0x00007FF755E25F9F] (No symbol) [0x00007FF755E20EC4] (No symbol) [0x00007FF755E21052] (No symbol) [0x00007FF755E118A4] BaseThreadInitThunk [0x00007FF98A42257D+29] RtlUserThreadStart [0x00007FF98B34AA48+40]
程序提示所有条目已处理并保存,但实际运行报错。此前尝试过UIPath也未能解决该问题,请问我哪里出错了?
问题分析与修复方案
核心错误点
- Driver被提前销毁:
fetch_images的finally块里调用driver.quit(),但main只初始化一次driver,第一次调用后driver就被关闭,后续处理其他标题时driver已失效,直接报错。 - 元素定位器失效:亚马逊页面结构可能更新,原商品链接选择器
a.a-link-normal.s-link-style无法匹配有效链接;且仅等待main-image-container存在,不代表内部图片已加载完成。 - 无头模式反爬检测:无头模式下亚马逊易识别为爬虫,导致页面渲染异常、元素无法定位。
- 固定sleep稳定性差:
time.sleep(5)无法适配不同网络速度,可能导致元素未加载完成就执行后续操作。
修复后的代码
import pandas as pd from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.common.keys import Keys from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time def setup_driver(): """Setup Chrome WebDriver with anti-detection options.""" options = Options() # 禁用无头模式(或添加参数模拟真实浏览器) # options.headless = True options.add_argument("--start-maximized") options.add_argument("--disable-blink-features=AutomationControlled") options.add_experimental_option("excludeSwitches", ["enable-automation"]) options.add_experimental_option('useAutomationExtension', False) service = Service(executable_path=r'C:\Users\preri\Desktop\Yorkee\chromedriver.exe') driver = webdriver.Chrome(service=service, options=options) # 修改navigator属性规避反爬检测 driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})") return driver def fetch_images(title, driver): """Fetch image URLs for a given product title from Amazon.""" try: driver.get("https://www.amazon.in") # 等待搜索框可点击并清空输入 search_box = WebDriverWait(driver, 10).until( EC.element_to_be_clickable((By.NAME, "field-keywords")) ) search_box.clear() search_box.send_keys(title + Keys.RETURN) # 等待商品列表加载,使用更准确的链接选择器 product_links = WebDriverWait(driver, 15).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, 'a.a-link-normal.s-no-outline')) ) if product_links: # 新标签页打开商品,避免页面跳转混乱 driver.execute_script("window.open('');") driver.switch_to.window(driver.window_handles[1]) driver.get(product_links[0].get_attribute('href')) # 等待图片元素加载完成 WebDriverWait(driver, 15).until( EC.presence_of_all_elements_located((By.CSS_SELECTOR, '#main-image-container .imgTagWrapper img')) ) # 提取有效图片URL images = driver.find_elements(By.CSS_SELECTOR, '#main-image-container .imgTagWrapper img') img_urls = [img.get_attribute('src') for img in images if img.get_attribute('src')] # 关闭新标签页并切回原页面 driver.close() driver.switch_to.window(driver.window_handles[0]) return img_urls else: print(f"No product links found for {title}") return [] except Exception as e: print(f"Error fetching images for {title}: {str(e)}") # 出错时恢复标签页状态 if len(driver.window_handles) > 1: driver.close() driver.switch_to.window(driver.window_handles[0]) return [] def main(): driver = setup_driver() df = pd.read_csv(r'C:\Users\preri\Desktop\Yorkee\Yorkee Sample Prod.csv') df['Title'] = df['Title'].astype(str) # 遍历处理每个标题,避免apply的隐式循环问题 df['Image URLs'] = [fetch_images(title, driver) for title in df['Title']] df.to_csv('modified_file_with_images.csv', index=False) print("All entries processed and saved.") # 所有任务完成后再关闭driver driver.quit() if __name__ == "__main__": main()
关键修复说明
- Driver生命周期优化:将
driver.quit()移至main函数末尾,确保所有标题处理完成后再销毁driver,避免提前关闭导致后续请求失败。 - 页面跳转处理:使用新标签页打开商品链接,处理完成后关闭并切回原标签页,防止页面跳转后元素定位混乱。
- 反爬增强:添加禁用自动化检测的参数,修改
navigator.webdriver属性,降低被亚马逊识别为爬虫的概率。 - 定位器更新:更换商品链接选择器为
a.a-link-normal.s-no-outline,适配亚马逊当前页面结构。 - 等待逻辑优化:用显式等待替代固定sleep,确保元素加载完成后再执行操作,提升稳定性。
- 错误处理完善:出错时自动恢复标签页状态,避免driver停留在错误页面影响后续任务。
内容的提问来源于stack exchange,提问作者Prerit Khanna
相关产品推荐
相关产品推荐

