You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

无法下载商品图片求助:获取标签内第二张图片链接并保存至DataFrame

解决Shopee商品第二张图片下载及链接存储问题

方案一:下载第二张商品图片

修改原有代码,定位页面商品图片列表、选取第二张后下载到本地:

import pandas as pd
import urllib.request
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

# 初始化浏览器
options = webdriver.ChromeOptions()
options.add_experimental_option('excludeSwitches', ['enable-logging'])
driver = webdriver.Chrome(options=options)
driver.maximize_window()

# 读取Excel链接表
df = pd.read_excel(r"C:\__Imagens e Planilhas Python\Shopee\Videos\Videos.xlsx")

# 遍历每个商品链接
for index, row in df.iterrows():
    driver.get(str(row["links"]))
    try:
        # 等待图片加载完成,定位所有商品图片
        image_elements = WebDriverWait(driver, 10).until(
            EC.presence_of_all_elements_located((By.CLASS_NAME, "_396QIb"))
        )
        # 取第二张图片(列表索引从0开始,所以用1)
        if len(image_elements) >= 2:
            second_image_src = image_elements[1].get_attribute("src")
            # 处理懒加载情况,若src为空则取data-src
            if not second_image_src:
                second_image_src = image_elements[1].get_attribute("data-src")
            
            # 下载图片到指定路径
            save_path = fr"C:\__Imagens e Planilhas Python\Shopee\Imagens Baixadas\{row['salvar']}.jpg"
            urllib.request.urlretrieve(second_image_src, save_path)
            print(f"已下载图片:{save_path}")
        else:
            print(f"商品链接 {row['links']} 图片数量不足2张")
    except Exception as e:
        print(f"处理商品链接 {row['links']} 出错:{str(e)}")

# 关闭浏览器
driver.quit()

方案二:将第二张图片链接保存到DataFrame

仅提取图片链接存入DataFrame,再导出到Excel:

import pandas as pd
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

# 初始化浏览器
options = webdriver.ChromeOptions()
options.add_experimental_option('excludeSwitches', ['enable-logging'])
driver = webdriver.Chrome(options=options)
driver.maximize_window()

# 读取Excel链接表,新增存储图片链接的列
df = pd.read_excel(r"C:\__Imagens e Planilhas Python\Shopee\Videos\Videos.xlsx")
df["image_link"] = ""

# 遍历每个商品链接
for index, row in df.iterrows():
    driver.get(str(row["links"]))
    try:
        # 等待图片加载完成,定位所有商品图片
        image_elements = WebDriverWait(driver, 10).until(
            EC.presence_of_all_elements_located((By.CLASS_NAME, "_396QIb"))
        )
        # 取第二张图片链接
        if len(image_elements) >= 2:
            second_image_src = image_elements[1].get_attribute("src")
            if not second_image_src:
                second_image_src = image_elements[1].get_attribute("data-src")
            df.at[index, "image_link"] = second_image_src
            print(f"已获取图片链接:{second_image_src}")
        else:
            df.at[index, "image_link"] = "图片数量不足2张"
            print(f"商品链接 {row['links']} 图片数量不足2张")
    except Exception as e:
        df.at[index, "image_link"] = f"获取失败:{str(e)}"
        print(f"处理商品链接 {row['links']} 出错:{str(e)}")

# 保存更新后的Excel文件
df.to_excel(r"C:\__Imagens e Planilhas Python\Shopee\Videos\Videos_com_links_imagens.xlsx", index=False)
print("已保存带图片链接的Excel文件")

# 关闭浏览器
driver.quit()

注意事项

  • 图片元素的class名_396QIb是Shopee商品页面的通用类名,若页面结构变更,需通过浏览器开发者工具重新定位元素选择器
  • 用WebDriverWait替代固定sleep,能更精准等待元素加载,避免无效等待
  • 针对懒加载图片,优先读取src属性,为空时再读取data-src属性

内容的提问来源于stack exchange,提问作者Felipe

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.18 06:30:55