You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何通过Selenium在非滚动页面(Airtable)中滚动表格?

解决Airtable嵌入表格无法滚动获取全部数据的问题

我需要解析某个Airtable嵌入页面的全部表格数据,但遇到了表格无法滚动的问题。已经尝试在浏览器控制台用JS滚动无效,也参考过相关滚动方案但没解决。下面是我尝试过的Python代码,包含多种滚动写法:

import requests
from bs4 import BeautifulSoup
from selenium import webdriver
import time
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver import ActionChains
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.common.keys import Keys

def get_html(url):
    service = Service(executable_path='C:\\airtable_parse\\chromedriver-win64\\chromedriver.exe')
    options = webdriver.ChromeOptions()
    driver = webdriver.Chrome(service=service, options=options)
    driver.maximize_window()

    try:
        driver.get(url=url)
        time.sleep(10)

        element = driver.find_element(By.XPATH, '//*[@id="table"]').send_keys(Keys.PAGE_DOWN) #Message: element not interactable
        ActionChains(driver).move_to_element(element).perform()

        #driver.execute_script("document.querySelector('#table').scrollTop = 500")

        # table = driver.find_element(By.XPATH, '//div[@id="firstContainer"]')
        # print(table)
        # action = ActionChains(driver)
        # action.move_to_element(table).perform()
        # print(action)
        # driver.execute_script('window.scrollTo(0, 300)', table)

        # while True:
        #     # After your page is loaded
        #     page_hight = driver.get_window_size()['height']  # Get page height
        #     scroll_bar = driver.find_element(By.XPATH, "//div[contains(@class,'antiscroll-scrollbar-vertical')]")
        #     ActionChains(driver).drag_and_drop_by_offset(scroll_bar, 0, page_hight - 160).click().perform()
        # #driver.execute_script('window.scrollBy(0, 100);')
    except Exception as e:
        print(e)
    finally:
        driver.close()
        driver.quit()


def main():
    get_html("https://airtable.com/embed/appImi8PX0i84XFwj/shr1PWZhR25O6DJxK/tblJG95RoC1WrRppF/viwVAi9l6dxxBtihM")


if __name__ == "__main__":
    main()

另外我还在控制台试过这些JS代码,也没效果:
document.querySelector("#view").scrollTop=300
document.querySelector("#viewContainer").scrollTop=300
document.querySelector("#table")
我觉得问题出在没有准确定位到负责滚动的HTML元素上。


解决方案:定位正确滚动容器并实现自动滚动

Airtable嵌入表格的实际滚动容器并非你尝试的#table,而是内部带特定样式的容器。以下是修正后的代码,可实现自动滚动加载全部数据:

import time
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.support.wait import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC

def get_html(url):
    service = Service(executable_path='C:\\airtable_parse\\chromedriver-win64\\chromedriver.exe')
    options = webdriver.ChromeOptions()
    driver = webdriver.Chrome(service=service, options=options)
    driver.maximize_window()

    try:
        driver.get(url)
        # 显式等待滚动容器加载完成,避免过早操作
        wait = WebDriverWait(driver, 20)
        scroll_container = wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, ".antiscroll-inner")))
        
        # 循环滚动直到没有新内容加载
        last_scroll_height = driver.execute_script("return arguments[0].scrollHeight", scroll_container)
        while True:
            # 滚动到容器底部
            driver.execute_script("arguments[0].scrollTop = arguments[0].scrollHeight", scroll_container)
            time.sleep(3)  # 等待新内容加载,可根据网络情况调整
            
            current_scroll_height = driver.execute_script("return arguments[0].scrollHeight", scroll_container)
            if current_scroll_height == last_scroll_height:
                break  # 高度不变说明已加载完所有数据
            last_scroll_height = current_scroll_height
        
        # 此时页面已加载全部数据,可获取源码进行解析
        page_source = driver.page_source
        # 这里添加BeautifulSoup解析逻辑,例如:
        # soup = BeautifulSoup(page_source, 'html.parser')
        # 提取表格数据...
        print("已加载全部数据,可开始解析")
        
    except Exception as e:
        print(f"错误信息: {e}")
    finally:
        driver.quit()

def main():
    get_html("https://airtable.com/embed/appImi8PX0i84XFwj/shr1PWZhR25O6DJxK/tblJG95RoC1WrRppF/viwVAi9l6dxxBtihM")

if __name__ == "__main__":
    main()

关键要点

  1. 滚动容器定位:Airtable嵌入视图的滚动容器通常是.antiscroll-inner,如果该选择器无效,可通过浏览器开发者工具查看元素的overflow属性(值为auto或scroll的元素即为滚动容器)
  2. 滚动逻辑:通过对比滚动前后容器的总高度,判断是否已加载完所有内容,避免无限循环
  3. 等待优化:用显式等待替代固定time.sleep,确保元素加载完成后再操作,提升代码稳定性

内容的提问来源于stack exchange,提问作者Konstantin

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.30 14:19:57