使用Python+Selenium抓取动态加载鞋类页面失败的技术求助
问题:Python+Selenium抓取theline.cl鞋类商品数据失败
我正在使用Python和Selenium抓取网页获取鞋类商品数据,目标页面为:https://theline.cl/categoria/zapatillas?page=15。尝试过requests库及多种等待方法均未成功,以下是我的代码(包含一些尝试后无效的写法):
def theline_scraper(): count = 0 page = 1 response = {} user_agent = 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/61.0.1750.517 Safari/537.36' options = Options() options.add_argument('user-agent={0}'.format(user_agent)) options.add_argument('start-maximized') # options.add_argument('disable-infobars') options.add_argument("--disable-extensions") options.add_argument("--headless") # Runs Chrome in headless mode. options.add_argument('--no-sandbox') # Bypass OS security model options.add_argument('--disable-dev-shm-usage') driver = webdriver.Chrome(options = options) START = 0 count = 0 response = {} driver.get("https://theline.cl/categoria/zapatillas?page={page}") driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") # Espera un momento para que la página se actualice time.sleep(10) wait = WebDriverWait(driver, 10) #elements = wait.until(EC.presence_of_all_elements_located((By.CLASS_NAME, 'mh-product-info'))) html = driver.page_source soup = BeautifulSoup(html, "html.parser") print(soup) zapatillas = [] for product in soup.find_all("div", {"class": "products__item"}): name = product.find("h3", {"class": "card-product__title"}).text.strip() price = product.find("span", {"class": "price--normal"}).text.strip() link = product.find('a')['href'] zapatilla = {"name": name, "price": price, "link": link} zapatillas.append(zapatilla) driver.quit() print(zapatillas) # for i in range(2): # THE_LINE_LINK = "https://theline.cl/categoria/zapatillas?page={page}" # while(True): # # try: # driver.get(THE_LINE_LINK) # html = driver.page_source # soup = BeautifulSoup(html, "html.parser") # wait = WebDriverWait(driver, 25).until(EC.presence_of_element_located((By.CLASS_NAME, 'mh-product-info'))) # print(wait.text) # break # # except: # # continue # html = driver.page_source # # Use BeautifulSoup to parse the HTML source code # soup = BeautifulSoup(html, "html.parser") # products = soup.find_all(class_='card-product') # print(products) # page += 1 # break return response
每次请求后得到的HTML都是可视化的页面内容(对应原问题中提及的图片内容)。
内容的提问来源于stack exchange,提问作者Diego Collao Alcayaga
相关产品推荐
相关产品推荐

