使用Selenium/Beautiful Soup无法抓取hulkapps-table表格的问题
网页表格抓取问题求助
问题背景
尝试抓取https://papemelroti.com/products/live-free-badge页面中的批量折扣表格,但始终无法定位到目标表格。目标表格的HTML结构如下:
<table class="hulkapps-table table"> <thead> <tr> <th style="border-top-left-radius: 0px;">Quantity</th> <th style="border-top-right-radius: 0px;">Bulk Discount</th> <th style="display: none">Add to Cart</th> </tr> </thead> <tbody> <tr> <td style="border-bottom-left-radius: 0px;">Buy 50 + <span class="hulk-offer-text"></span></td> <td style="border-bottom-right-radius: 0px;"><span class="hulkapps-price"><span class="money"><span class="money"> ₱1.00 </span></span> Off</span></td> <td style="display: none;"><button type="button" class="AddToCart_0" style="cursor: pointer; font-weight: 600; letter-spacing: .08em; font-size: 11px; padding: 5px 15px; border-color: #171515; border-width: 2px; color: #ffffff; background: #161212;" onclick="add_to_cart(50)">Add to Cart</button></td> </tr> </tbody> </table>
我的代码
编写了Selenium+BeautifulSoup的抓取代码,但运行后始终返回None,无法获取表格内容:
from selenium import webdriver from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from bs4 import BeautifulSoup import time # Set up Chrome options chrome_options = Options() chrome_options.add_argument("--headless") chrome_options.add_argument("--no-sandbox") chrome_options.add_argument("--disable-dev-shm-usage") service = Service('/usr/local/bin/chromedriver') # Adjust path if necessary driver = webdriver.Chrome(service=service, options=chrome_options) def get_page_html(url): driver.get(url) time.sleep(3) # Wait for JS to load return driver.page_source def scrape_discount_quantity(url): page_html = get_page_html(url) soup = BeautifulSoup(page_html, "html.parser") # Locate the table containing the quantity and discount table = soup.find('table', class_='hulkapps-table') print(page_html) if table: table_rows = table.find_all('tr') for row in table_rows: quantity_cells = row.find_all('td') if len(quantity_cells) >= 2: # Check if there are at least two cells quantity_cell = quantity_cells[0].get_text(strip=True) # Get quantity text discount_cell = quantity_cells[1].get_text(strip=True) # Get discount text return quantity_cell, discount_cell return None, None # Example usage url = 'https://papemelroti.com/products/live-free-badge' quantity, discount = scrape_discount_quantity(url) print(f"Quantity: {quantity}, Discount: {discount}") driver.quit() # Close the browser when done
问题分析与解决方案
核心问题
- 固定等待时间不足:
time.sleep(3)无法保证动态表格加载完成,页面可能需要更长时间渲染内容。 - 表格定位与遍历效率低:原代码遍历所有
<tr>(包括表头的行),且定位表格的方式可以更精准。
修正后的代码(方案1:优化等待+精准定位)
使用Selenium的显式等待替代固定sleep,确保表格加载完成后再解析;同时用CSS选择器精准匹配多类表格:
from selenium import webdriver from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC from bs4 import BeautifulSoup # Set up Chrome options chrome_options = Options() chrome_options.add_argument("--headless") chrome_options.add_argument("--no-sandbox") chrome_options.add_argument("--disable-dev-shm-usage") service = Service('/usr/local/bin/chromedriver') # Adjust path if necessary driver = webdriver.Chrome(service=service, options=chrome_options) def get_page_html(url): driver.get(url) # 等待表格出现,最长等待10秒 WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CSS_SELECTOR, "table.hulkapps-table.table")) ) return driver.page_source def scrape_discount_quantity(url): page_html = get_page_html(url) soup = BeautifulSoup(page_html, "html.parser") # 用CSS选择器定位多类表格 table = soup.select_one('table.hulkapps-table.table') if table: # 直接遍历tbody中的数据行,跳过表头 tbody_rows = table.find('tbody').find_all('tr') for row in tbody_rows: cells = row.find_all('td') if len(cells) >= 2: quantity = cells[0].get_text(strip=True) discount = cells[1].get_text(strip=True) return quantity, discount return None, None # Example usage url = 'https://papemelroti.com/products/live-free-badge' quantity, discount = scrape_discount_quantity(url) print(f"Quantity: {quantity}, Discount: {discount}") driver.quit()
修正后的代码(方案2:直接用Selenium定位,无需BeautifulSoup)
省去HTML解析步骤,直接用Selenium定位表格元素,更高效:
from selenium import webdriver from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC # Set up Chrome options chrome_options = Options() chrome_options.add_argument("--headless") chrome_options.add_argument("--no-sandbox") chrome_options.add_argument("--disable-dev-shm-usage") service = Service('/usr/local/bin/chromedriver') # Adjust path if necessary driver = webdriver.Chrome(service=service, options=chrome_options) def scrape_discount_quantity(url): driver.get(url) try: # 等待表格加载完成 table = WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.CLASS_NAME, "hulkapps-table")) ) # 获取tbody中的数据行 rows = table.find_element(By.TAG_NAME, "tbody").find_elements(By.TAG_NAME, "tr") for row in rows: cells = row.find_elements(By.TAG_NAME, "td") if len(cells) >= 2: quantity = cells[0].text.strip() discount = cells[1].text.strip() return quantity, discount except Exception as e: print(f"抓取失败: {e}") return None, None return None, None # Example usage url = 'https://papemelroti.com/products/live-free-badge' quantity, discount = scrape_discount_quantity(url) print(f"Quantity: {quantity}, Discount: {discount}") driver.quit()
内容的提问来源于stack exchange,提问作者Rav
相关产品推荐
相关产品推荐

