Selenium爬取Hepsiburada笔记本:如何提取产品名称与价格
问题
使用Python Selenium编写了爬取Hepsiburada联想笔记本页面的代码,已成功获取页面上的笔记本WebElement列表,但无法定位每个笔记本的产品名称和价格信息,尝试使用XPATH、CSS_SELECTOR、CLASS_NAME均失败,请问该如何解决?
用户提供的代码:
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.firefox.options import Options from selenium.webdriver.common.action_chains import ActionChains import time def scrape_laptops(driver): # Laptop listesini içeren üst elementi bul laptop_list = driver.find_element(By.XPATH, '//*[@id="1"]') # Liste içindeki tüm laptopları bul laptops = laptop_list.find_elements(By.TAG_NAME, 'li') print(laptops) laptop_data = [] for laptop in laptops: # Her laptop için model ve fiyat bilgisini çek model = laptop.find_elements(By.XPATH, '/html/body/div[1]/div/div/main/div/div/div[2]/div/div[2]/div[3]/div/div[2]/div/div/div/div[2]/div/ul[1]/li[1]/div/a/div[2]/div[2]/h3') for mod in model: print(mod) print(model) price = laptop.find_element(By.XPATH, '/html/body/div[1]/div/div/main/div/div/div[2]/div/div[2]/div[3]/div/div[2]/div/div/div/div[2]/div/ul[1]/li[1]/div/a/div[2]/div[4]') laptop_data.append({'model': model, 'price': price}) print(price) print(laptop) return laptop_data # Firefox options options = Options() options.headless = True # Start Firefox webdriver driver = webdriver.Firefox(options=options) # Go to the Hepsiburada Lenovo computers page driver.get('https://www.hepsiburada.com/lenovo/bilgisayarlar-c-3000500') # Accept cookies button2 = WebDriverWait(driver, 20).until( EC.element_to_be_clickable((By.ID, "onetrust-accept-btn-handler")) ) button2.click() # Wait for the page to load time.sleep(5) # Scroll to the bottom of the page driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(5) # Click the "Show more" button button = WebDriverWait(driver, 20).until( EC.element_to_be_clickable((By.XPATH, '/html/body/div[1]/div/div/main/div[1]/div/div[2]/div/div[2]/div[3]/div/div[2]/div/div/div/div/div/div/div[1]/button')) ) actions = ActionChains(driver) actions.move_to_element(button).perform() button.click() time.sleep(5) # Scrape laptop models and prices laptop_data = scrape_laptops(driver) # Write laptop data to a file with open("laptop_prices.txt", "w", encoding="utf-8") as file: for laptop in laptop_data: file.write(f"Model: {laptop['model']}, Fiyat: {laptop['price']}\n") # Close the webdriver driver.quit()
解决方案
你的核心问题是绝对XPATH滥用和未利用元素上下文定位,以下是具体修复方案:
问题分析
- 绝对XPATH从根节点开始定位,页面结构稍有变动就会失效;且遍历单个笔记本元素时,没有限定查找范围,导致Selenium始终在整个页面搜索,而非当前笔记本元素内部。
- 直接存储WebElement对象而非提取文本,最终写入文件的是对象引用,不是实际的产品名称和价格。
修复步骤
- 改用相对XPATH:在单个
laptop元素内,使用.开头的相对路径定位子元素,确保查找范围限定在当前笔记本元素中。 - 提取元素文本:通过
.text属性获取产品名称和价格的实际内容。 - 优化等待策略:用显式等待替代部分
time.sleep(),提升代码稳定性。
修改后的完整代码
from selenium import webdriver from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from selenium.webdriver.firefox.options import Options from selenium.webdriver.common.action_chains import ActionChains import time def scrape_laptops(driver): # 等待笔记本列表加载完成 laptop_list = WebDriverWait(driver, 20).until( EC.presence_of_element_located((By.XPATH, '//*[@id="1"]')) ) # 获取所有笔记本元素 laptops = laptop_list.find_elements(By.TAG_NAME, 'li') laptop_data = [] for laptop in laptops: try: # 相对路径定位产品名称(当前laptop元素内查找) model_elem = WebDriverWait(laptop, 10).until( EC.presence_of_element_located((By.XPATH, './/h3[contains(@class, "product-name")]')) ) model = model_elem.text.strip() # 相对路径定位价格元素 price_elem = WebDriverWait(laptop, 10).until( EC.presence_of_element_located((By.XPATH, './/div[contains(@class, "price")]')) ) price = price_elem.text.strip() laptop_data.append({'model': model, 'price': price}) print(f"Model: {model}, Fiyat: {price}") except Exception as e: print(f"获取笔记本信息失败: {str(e)}") continue return laptop_data # Firefox配置 options = Options() options.headless = True # 启动浏览器 driver = webdriver.Firefox(options=options) driver.get('https://www.hepsiburada.com/lenovo/bilgisayarlar-c-3000500') # 接受Cookie WebDriverWait(driver, 20).until( EC.element_to_be_clickable((By.ID, "onetrust-accept-btn-handler")) ).click() # 滚动到底部加载更多内容 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(3) # 点击"显示更多"按钮 show_more_btn = WebDriverWait(driver, 20).until( EC.element_to_be_clickable((By.XPATH, './/button[contains(text(), "Daha Fazla Göster")]')) ) ActionChains(driver).move_to_element(show_more_btn).click().perform() time.sleep(3) # 爬取数据 laptop_data = scrape_laptops(driver) # 写入文件 with open("laptop_prices.txt", "w", encoding="utf-8") as file: for laptop in laptop_data: file.write(f"Model: {laptop['model']}, Fiyat: {laptop['price']}\n") # 关闭浏览器 driver.quit()
关键优化点
- 使用
WebDriverWait确保元素加载完成后再操作,避免因页面未加载完全导致的定位失败。 - 用
contains(@class, "xxx")匹配类名,避免因元素多类名导致的定位失效。 - 添加异常处理,跳过加载失败的笔记本元素,防止程序崩溃。
内容的提问来源于stack exchange,提问作者Berkan Özelmacı
相关产品推荐
相关产品推荐

