You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Playwright+Selectolax爬虫仅爬至aria-rowindex=29,需爬至2509求助

问题解决思路及代码优化

核心问题分析

  • 原代码依赖固定时间等待(time.sleep()),无法适配页面动态加载的实际耗时
  • 直接滚动到页面底部的方式,对虚拟滚动表格无效(这类表格只会渲染可视区域内的行,旧行会被DOM移除)
  • 停止加载的判断逻辑过于激进,可能在新内容未完全加载时就终止滚动

优化后的代码

from playwright.sync_api import sync_playwright
from selectolax.parser import HTMLParser
import pandas as pd

def extract_full_body_html(url):
    TIMEOUT = 60000  # 延长超时时间适配大列表加载

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=False)  # 先关闭无头模式调试,稳定后再开启
        page = browser.new_page(viewport={'width': 1920, 'height': 1080})
        
        page.goto(url, wait_until='networkidle')
        # 等待表格容器加载
        page.wait_for_selector('div[role="grid"]', timeout=TIMEOUT)

        def load_more_content():
            target_row_index = 2509
            last_count = 0
            
            while True:
                # 获取当前页面中最大的aria-rowindex
                current_max_index = page.evaluate('''() => {
                    const cells = document.querySelectorAll('div[role="gridcell"][aria-rowindex]');
                    if (cells.length === 0) return 0;
                    return Math.max(...Array.from(cells).map(c => parseInt(c.getAttribute("aria-rowindex"))));
                }''')

                # 已经达到目标,停止滚动
                if current_max_index >= target_row_index:
                    break

                # 滚动到最后一个可见的行元素,触发虚拟滚动加载
                last_row = page.query_selector('div[role="gridcell"][aria-rowindex="{}"]'.format(current_max_index))
                if last_row:
                    last_row.scroll_into_view_if_needed()
                
                # 等待新元素加载,最多等10秒
                try:
                    page.wait_for_function(
                        '''(last_count) => document.querySelectorAll('div[role="gridcell"][aria-rowindex]').length > last_count''',
                        last_count,
                        timeout=10000
                    )
                    last_count = page.evaluate('''() => document.querySelectorAll('div[role="gridcell"][aria-rowindex]').length''')
                except:
                    # 超过10秒没有新元素加载,判定为加载完成
                    break

        load_more_content()
        # 获取整个表格的HTML,而不是body,避免无关内容干扰
        table_html = page.inner_html('div[role="grid"]')
        browser.close()
        return table_html

def extraction(html):
    tree = HTMLParser(html)
    data = []

    # 先获取所有存在的row index,避免循环中断
    row_indices = set()
    for cell in tree.css('div[role="gridcell"][aria-rowindex]'):
        idx = cell.attributes.get('aria-rowindex')
        if idx:
            row_indices.add(int(idx))
    
    # 按顺序遍历所有存在的index,直到2509
    for i in sorted(row_indices):
        if i > 2509:
            break
        
        row_selector = f'div[role="gridcell"][aria-rowindex="{i}"]'
        company_div = tree.css_first(f'{row_selector}[aria-colindex="1"]')
        if not company_div:
            continue
        
        row_data = {
            'Company': company_div.text(deep=True, separator=' '),
            'Emails': tree.css_first(f'{row_selector}[aria-colindex="2"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="2"]') else '',
            'Addresses': tree.css_first(f'{row_selector}[aria-colindex="3"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="3"]') else '',
            'Urls': tree.css_first(f'{row_selector}[aria-colindex="4"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="4"]') else '',
            'Description': tree.css_first(f'{row_selector}[aria-colindex="5"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="5"]') else '',
            'Stage': tree.css_first(f'{row_selector}[aria-colindex="6"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="6"]') else '',
            'Number of Portfolio Organizations': tree.css_first(f'{row_selector}[aria-colindex="7"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="7"]') else '',
            'Number of Investments': tree.css_first(f'{row_selector}[aria-colindex="8"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="8"]') else '',
            'Accelerator Duration (in weeks)': tree.css_first(f'{row_selector}[aria-colindex="9"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="9"]') else '',
            'Number of Exits': tree.css_first(f'{row_selector}[aria-colindex="10"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="10"]') else '',
            'Linkedin': tree.css_first(f'{row_selector}[aria-colindex="11"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="11"]') else '',
            'Founders': tree.css_first(f'{row_selector}[aria-colindex="12"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="12"]') else '',
            'Twitter': tree.css_first(f'{row_selector}[aria-colindex="13"]').text(deep=True, separator=' ') if tree.css_first(f'{row_selector}[aria-colindex="13"]') else ''
        }
        data.append(row_data)

    return data

if __name__ == '__main__':
    url = 'https://app.folk.app/shared/All-accelerators-rw0kuUNqtzl6j6dDQquoZTYF6MFKIQHo'
    html = extract_full_body_html(url)
    data = extraction(html)
    df = pd.DataFrame(data)
    df.to_excel('output.xlsx', index=False)

关键修改说明

  • 滚动逻辑优化:不再滚动到页面底部,而是滚动到当前最大索引的行元素,触发虚拟滚动的加载机制,适配大多数现代表格的渲染方式
  • 替换固定等待:用wait_for_function等待新元素出现,替代time.sleep(),避免不必要的等待或等待不足
  • 调整停止条件:直接判断当前最大索引是否达到目标值(2509),同时增加超时容错,超过10秒无新元素则停止
  • 提取逻辑优化:先收集所有存在的row索引,再按顺序遍历,避免因虚拟滚动移除旧行导致的循环中断;同时为每个字段增加空值判断,防止报错
  • 调试友好:默认关闭无头模式,方便观察页面加载情况,稳定后可改为headless=True

内容的提问来源于stack exchange,提问作者Muhammad

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.27 00:54:57