Python Web Scraping问题:hoverable menus失效与懒加载图片未加载
问题描述
尝试对ClassCentral网站进行深度1级的网页爬取,爬取完成后发现页面中的悬浮菜单无法正常工作,大量图片仍处于模糊状态。该网站采用滚动加载机制,仅当滚动到图片位置并等待后才会加载清晰图片,我尝试模拟该滚动加载过程但未成功,不清楚问题出在哪里。
我的代码
from selenium import webdriver from urllib.parse import urljoin import os from selenium.webdriver.common.by import By from selenium.common.exceptions import StaleElementReferenceException from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time def scroll_to_bottom(driver): """ Scrolls the page to the bottom slowly. """ SCROLL_PAUSE_TIME = 1 MAX_SCROLLS = 10 SCROLL_INCREMENT = driver.execute_script("return window.innerHeight;") last_height = driver.execute_script("return document.body.scrollHeight") scrolls = 0 while scrolls < MAX_SCROLLS: driver.execute_script(f"window.scrollBy(0, {SCROLL_INCREMENT});") time.sleep(SCROLL_PAUSE_TIME) new_height = driver.execute_script("return document.body.scrollHeight") if new_height == last_height: break last_height = new_height scrolls += 1 def scrape_page(driver, url, directory, visited): """ Scrapes a single page and saves it to the specified directory. """ try: driver.get(url) scroll_to_bottom(driver) WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) with open(f'{directory}/{url.replace("/", "-").replace("https:--www.classcentral.com-", "")}.html', 'w', encoding='utf-8') as f: f.write(driver.page_source) print(f"Scraped {url}") sub_links = driver.find_elements(By.CSS_SELECTOR, 'a') for sub_link in sub_links: try: sub_url = sub_link.get_attribute('href') if sub_url and 'classcentral.com' in sub_url and sub_url != url and sub_url not in visited: sub_url = urljoin('https://www.classcentral.com/', sub_url) visited.add(sub_url) driver.get(sub_url) scroll_to_bottom(driver) scrape_page(driver, sub_url, directory, visited) except StaleElementReferenceException: continue time.sleep(10) except StaleElementReferenceException: pass def scrape_site(url, max_depth): print("Scraping site...") """ Scrapes the specified site up to a maximum depth. """ driver = webdriver.Chrome() driver.get(url) WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) page_title = driver.title directory = f'classcentral-{page_title}' if not os.path.exists(directory): os.makedirs(directory) visited = {url} print(f"Visited: {visited}") scrape_page(driver, url, directory, visited) queue = [(url, 1)] while queue: current_url, current_depth = queue.pop(0) if current_depth > max_depth: break driver.get(current_url) links = driver.find_elements(By.CSS_SELECTOR, 'a, script[src]') for link in links: try: url = link.get_attribute('href') or link.get_attribute('src') if url and 'classcentral.com' in url and url not in visited and not url.startswith('mailto:'): visited.add(url) queue.append((url, current_depth + 1)) url = urljoin('https://www.classcentral.com/', url) scrape_page(driver, url, directory, visited) except StaleElementReferenceException: continue driver.quit() scrape_site('https://www.classcentral.com/', 1)
问题排查与修复方案
1. 滚动加载失效的核心问题
原scroll_to_bottom函数存在两个关键缺陷:
- 滚动次数限制:
MAX_SCROLLS=10可能导致未滚动到页面底部,遗漏部分图片的加载触发; - 等待逻辑粗糙:固定
time.sleep(1)不足以让网站完成清晰图片的加载,且没有针对图片元素的加载状态做等待。
修复方式:改为逐个定位页面中的图片元素,滚动到每个图片位置并等待其加载完成,确保所有模糊占位图都被替换为清晰图:
def load_all_images(driver): # 获取所有图片元素 images = driver.find_elements(By.TAG_NAME, 'img') for img in images: try: # 滚动到图片位置 driver.execute_script("arguments[0].scrollIntoView({behavior: 'smooth', block: 'center'});", img) # 等待图片加载完成(判断naturalWidth是否大于0) WebDriverWait(driver, 5).until( lambda d: d.execute_script("return arguments[0].naturalWidth > 0", img) ) except Exception: continue # 最后滚动到底部确保没有遗漏 driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2)
2. 悬浮菜单失效问题
爬取保存的page_source仅包含当前DOM结构,不包含页面依赖的JS、CSS等外部资源,而悬浮菜单的交互完全依赖这些动态脚本。解决思路:
- 爬取页面时同步下载所有关联的JS、CSS文件,并修改HTML中资源的引用路径为本地路径;
- 仅保存静态HTML无法恢复交互功能,需额外处理静态资源的下载和路径替换。
3. 爬取逻辑冗余混乱
原代码同时使用递归scrape_page和队列queue处理页面,导致重复爬取和频繁页面切换,容易触发StaleElementReferenceException。修复方式:统一使用队列管理待爬取页面,避免递归调用:
def scrape_site(url, max_depth): print("Scraping site...") driver = webdriver.Chrome() driver.get(url) WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) page_title = driver.title directory = f'classcentral-{page_title}' if not os.path.exists(directory): os.makedirs(directory) visited = set() queue = [(url, 1)] while queue: current_url, current_depth = queue.pop(0) if current_depth > max_depth or current_url in visited: continue visited.add(current_url) try: driver.get(current_url) # 加载所有图片 load_all_images(driver) # 等待页面完全加载 WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) # 保存页面 filename = current_url.replace("https://www.classcentral.com/", "").replace("/", "-").strip("-") + ".html" with open(os.path.join(directory, filename), 'w', encoding='utf-8') as f: f.write(driver.page_source) print(f"Scraped {current_url}") # 收集深度内的子链接 if current_depth < max_depth: sub_links = driver.find_elements(By.CSS_SELECTOR, 'a') for sub_link in sub_links: try: sub_url = sub_link.get_attribute('href') if sub_url and 'classcentral.com' in sub_url and sub_url not in visited and not sub_url.startswith('mailto:'): sub_url = urljoin('https://www.classcentral.com/', sub_url) queue.append((sub_url, current_depth + 1)) except StaleElementReferenceException: continue time.sleep(2) except Exception as e: print(f"Failed to scrape {current_url}: {str(e)}") continue driver.quit()
完整修正代码
from selenium import webdriver from urllib.parse import urljoin import os from selenium.webdriver.common.by import By from selenium.common.exceptions import StaleElementReferenceException from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC import time def load_all_images(driver): """滚动到每个图片位置,等待清晰图片加载完成""" images = driver.find_elements(By.TAG_NAME, 'img') for img in images: try: driver.execute_script("arguments[0].scrollIntoView({behavior: 'smooth', block: 'center'});", img) WebDriverWait(driver, 5).until( lambda d: d.execute_script("return arguments[0].naturalWidth > 0", img) ) except Exception: continue driver.execute_script("window.scrollTo(0, document.body.scrollHeight);") time.sleep(2) def scrape_site(url, max_depth): print("Scraping site...") driver = webdriver.Chrome() driver.get(url) WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) page_title = driver.title directory = f'classcentral-{page_title}' if not os.path.exists(directory): os.makedirs(directory) visited = set() queue = [(url, 1)] while queue: current_url, current_depth = queue.pop(0) if current_depth > max_depth or current_url in visited: continue visited.add(current_url) try: driver.get(current_url) load_all_images(driver) WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body'))) # 生成安全的文件名 filename = current_url.replace("https://www.classcentral.com/", "").replace("/", "-").strip("-") + ".html" filepath = os.path.join(directory, filename) with open(filepath, 'w', encoding='utf-8') as f: f.write(driver.page_source) print(f"Scraped {current_url} -> {filepath}") # 收集子链接(仅当当前深度未达上限) if current_depth < max_depth: sub_links = driver.find_elements(By.CSS_SELECTOR, 'a') for sub_link in sub_links: try: sub_url = sub_link.get_attribute('href') if sub_url and 'classcentral.com' in sub_url and sub_url not in visited and not sub_url.startswith('mailto:'): sub_url = urljoin('https://www.classcentral.com/', sub_url) queue.append((sub_url, current_depth + 1)) except StaleElementReferenceException: continue time.sleep(2) except Exception as e: print(f"Failed to scrape {current_url}: {str(e)}") continue driver.quit() scrape_site('https://www.classcentral.com/', 1)
内容的提问来源于stack exchange,提问作者Benjji
相关产品推荐
相关产品推荐

