You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Python Web Scraping问题:hoverable menus失效与懒加载图片未加载

问题描述

尝试对ClassCentral网站进行深度1级的网页爬取,爬取完成后发现页面中的悬浮菜单无法正常工作,大量图片仍处于模糊状态。该网站采用滚动加载机制,仅当滚动到图片位置并等待后才会加载清晰图片,我尝试模拟该滚动加载过程但未成功,不清楚问题出在哪里。

我的代码
from selenium import webdriver
from urllib.parse import urljoin
import os
from selenium.webdriver.common.by import By
from selenium.common.exceptions import StaleElementReferenceException
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import time

def scroll_to_bottom(driver):
    """
    Scrolls the page to the bottom slowly.
    """
    SCROLL_PAUSE_TIME = 1
    MAX_SCROLLS = 10
    SCROLL_INCREMENT = driver.execute_script("return window.innerHeight;")
    last_height = driver.execute_script("return document.body.scrollHeight")
    scrolls = 0
    while scrolls < MAX_SCROLLS:
        driver.execute_script(f"window.scrollBy(0, {SCROLL_INCREMENT});")
        time.sleep(SCROLL_PAUSE_TIME)
        new_height = driver.execute_script("return document.body.scrollHeight")
        if new_height == last_height:
            break
        last_height = new_height
        scrolls += 1

def scrape_page(driver, url, directory, visited):
    """
    Scrapes a single page and saves it to the specified directory.
    """
    try:
        driver.get(url)
        scroll_to_bottom(driver)
        WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
        with open(f'{directory}/{url.replace("/", "-").replace("https:--www.classcentral.com-", "")}.html', 'w', encoding='utf-8') as f:
            f.write(driver.page_source)
            print(f"Scraped {url}")
            
        sub_links = driver.find_elements(By.CSS_SELECTOR, 'a')
        for sub_link in sub_links:
            try:
                sub_url = sub_link.get_attribute('href')
                if sub_url and 'classcentral.com' in sub_url and sub_url != url and sub_url not in visited:
                    sub_url = urljoin('https://www.classcentral.com/', sub_url)
                    visited.add(sub_url)
                    driver.get(sub_url)
                    scroll_to_bottom(driver)
                    scrape_page(driver, sub_url, directory, visited)
            except StaleElementReferenceException:
                continue
            time.sleep(10)
    except StaleElementReferenceException:
        pass

def scrape_site(url, max_depth):
    print("Scraping site...")
    """
    Scrapes the specified site up to a maximum depth.
    """
    driver = webdriver.Chrome()
    driver.get(url)
    WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
    page_title = driver.title
    directory = f'classcentral-{page_title}'
    if not os.path.exists(directory):
        os.makedirs(directory)
    
    visited = {url}
    print(f"Visited: {visited}")

    scrape_page(driver, url, directory, visited)
    
    queue = [(url, 1)]
    while queue:
        current_url, current_depth = queue.pop(0)
        if current_depth > max_depth:
            break
        driver.get(current_url)
        links = driver.find_elements(By.CSS_SELECTOR, 'a, script[src]')
        for link in links:
            try:
                url = link.get_attribute('href') or link.get_attribute('src')
                if url and 'classcentral.com' in url and url not in visited and not url.startswith('mailto:'):
                    visited.add(url)
                    queue.append((url, current_depth + 1))
                    url = urljoin('https://www.classcentral.com/', url)
                    
                    scrape_page(driver, url, directory, visited)
            except StaleElementReferenceException:
                
                continue
    
    driver.quit()

scrape_site('https://www.classcentral.com/', 1)
问题排查与修复方案

1. 滚动加载失效的核心问题

原scroll_to_bottom函数存在两个关键缺陷:

  • 滚动次数限制:MAX_SCROLLS=10可能导致未滚动到页面底部,遗漏部分图片的加载触发;
  • 等待逻辑粗糙:固定time.sleep(1)不足以让网站完成清晰图片的加载,且没有针对图片元素的加载状态做等待。

修复方式:改为逐个定位页面中的图片元素,滚动到每个图片位置并等待其加载完成,确保所有模糊占位图都被替换为清晰图:

def load_all_images(driver):
    # 获取所有图片元素
    images = driver.find_elements(By.TAG_NAME, 'img')
    for img in images:
        try:
            # 滚动到图片位置
            driver.execute_script("arguments[0].scrollIntoView({behavior: 'smooth', block: 'center'});", img)
            # 等待图片加载完成(判断naturalWidth是否大于0)
            WebDriverWait(driver, 5).until(
                lambda d: d.execute_script("return arguments[0].naturalWidth > 0", img)
            )
        except Exception:
            continue
    # 最后滚动到底部确保没有遗漏
    driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
    time.sleep(2)

2. 悬浮菜单失效问题

爬取保存的page_source仅包含当前DOM结构,不包含页面依赖的JS、CSS等外部资源,而悬浮菜单的交互完全依赖这些动态脚本。解决思路:

  • 爬取页面时同步下载所有关联的JS、CSS文件,并修改HTML中资源的引用路径为本地路径;
  • 仅保存静态HTML无法恢复交互功能,需额外处理静态资源的下载和路径替换。

3. 爬取逻辑冗余混乱

原代码同时使用递归scrape_page和队列queue处理页面,导致重复爬取和频繁页面切换,容易触发StaleElementReferenceException。修复方式:统一使用队列管理待爬取页面,避免递归调用:

def scrape_site(url, max_depth):
    print("Scraping site...")
    driver = webdriver.Chrome()
    driver.get(url)
    WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
    page_title = driver.title
    directory = f'classcentral-{page_title}'
    if not os.path.exists(directory):
        os.makedirs(directory)
    
    visited = set()
    queue = [(url, 1)]
    
    while queue:
        current_url, current_depth = queue.pop(0)
        if current_depth > max_depth or current_url in visited:
            continue
        visited.add(current_url)
        
        try:
            driver.get(current_url)
            # 加载所有图片
            load_all_images(driver)
            # 等待页面完全加载
            WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
            
            # 保存页面
            filename = current_url.replace("https://www.classcentral.com/", "").replace("/", "-").strip("-") + ".html"
            with open(os.path.join(directory, filename), 'w', encoding='utf-8') as f:
                f.write(driver.page_source)
            print(f"Scraped {current_url}")
            
            # 收集深度内的子链接
            if current_depth < max_depth:
                sub_links = driver.find_elements(By.CSS_SELECTOR, 'a')
                for sub_link in sub_links:
                    try:
                        sub_url = sub_link.get_attribute('href')
                        if sub_url and 'classcentral.com' in sub_url and sub_url not in visited and not sub_url.startswith('mailto:'):
                            sub_url = urljoin('https://www.classcentral.com/', sub_url)
                            queue.append((sub_url, current_depth + 1))
                    except StaleElementReferenceException:
                        continue
            time.sleep(2)
        except Exception as e:
            print(f"Failed to scrape {current_url}: {str(e)}")
            continue
    
    driver.quit()
完整修正代码
from selenium import webdriver
from urllib.parse import urljoin
import os
from selenium.webdriver.common.by import By
from selenium.common.exceptions import StaleElementReferenceException
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
import time

def load_all_images(driver):
    """滚动到每个图片位置,等待清晰图片加载完成"""
    images = driver.find_elements(By.TAG_NAME, 'img')
    for img in images:
        try:
            driver.execute_script("arguments[0].scrollIntoView({behavior: 'smooth', block: 'center'});", img)
            WebDriverWait(driver, 5).until(
                lambda d: d.execute_script("return arguments[0].naturalWidth > 0", img)
            )
        except Exception:
            continue
    driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
    time.sleep(2)

def scrape_site(url, max_depth):
    print("Scraping site...")
    driver = webdriver.Chrome()
    driver.get(url)
    WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
    page_title = driver.title
    directory = f'classcentral-{page_title}'
    if not os.path.exists(directory):
        os.makedirs(directory)
    
    visited = set()
    queue = [(url, 1)]
    
    while queue:
        current_url, current_depth = queue.pop(0)
        if current_depth > max_depth or current_url in visited:
            continue
        visited.add(current_url)
        
        try:
            driver.get(current_url)
            load_all_images(driver)
            WebDriverWait(driver, 40).until(EC.presence_of_element_located((By.TAG_NAME, 'body')))
            
            # 生成安全的文件名
            filename = current_url.replace("https://www.classcentral.com/", "").replace("/", "-").strip("-") + ".html"
            filepath = os.path.join(directory, filename)
            with open(filepath, 'w', encoding='utf-8') as f:
                f.write(driver.page_source)
            print(f"Scraped {current_url} -> {filepath}")
            
            # 收集子链接(仅当当前深度未达上限)
            if current_depth < max_depth:
                sub_links = driver.find_elements(By.CSS_SELECTOR, 'a')
                for sub_link in sub_links:
                    try:
                        sub_url = sub_link.get_attribute('href')
                        if sub_url and 'classcentral.com' in sub_url and sub_url not in visited and not sub_url.startswith('mailto:'):
                            sub_url = urljoin('https://www.classcentral.com/', sub_url)
                            queue.append((sub_url, current_depth + 1))
                    except StaleElementReferenceException:
                        continue
            time.sleep(2)
        except Exception as e:
            print(f"Failed to scrape {current_url}: {str(e)}")
            continue
    
    driver.quit()

scrape_site('https://www.classcentral.com/', 1)

内容的提问来源于stack exchange,提问作者Benjji

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.29 15:15:10