You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将Selenium集成到现有代码抓取全量参展商数据

问题

我需要从链接https://asiatechxsg.com/exhibitors/抓取所有参展商名称及信息并保存为CSV文件。目前我已编写了基于requests和BeautifulSoup的代码,但仅能抓取50条参展商数据,推测是页面动态加载导致。我不熟悉Selenium,请问如何将其集成到现有代码中以实现全量抓取?

现有代码如下:

html = requests.get('https://asiatechxsg.com/exhibitors/').text
bs = BeautifulSoup(html)
exhibitor_links = []
for link in bs.find_all('a'):
    if link.has_attr('href'):
        exhibitor_links.append(link.attrs['href'])
        print(link.attrs['href'])

exhibitor_names = []
exhibitor_info = []
exhibitor_web = []
exhibitor_linkedin = []
exhibitor_twitter = []
exhibitor_email = []
exhibitor_contact = []
exhibitor_booth = []

base_url = "https://attend.informatechevents.virtual.informatech.com/"
for link in exhibitor_links:
    url = base_url + link
    bs = BeautifulSoup(requests.get(url).text, 'html.parser')
 
    name = bs.find('h1', class_= "sc-c418aba9-7 dbnofQ").text
    info_div = bs.find('div', class_='sc-dd6f9f7c-0 hIHCKf')

    # Get exhibitor information
    if info_div:
        info_paras= info_div.find_all('p')
        info = '\n'.join([p.get_text() for p in info_paras])
        exhibitor_info.append(info)
    else:
        exhibitor_info.append("")
    exhibitor_names.append(name)

    # Get exhibitor website, contact, email
    tel = ""
    website = ""
    email = ""

    contact_links = bs.find_all('a', class_='sc-2d6f1d18-0 isnIyX')
    for a in contact_links:
        href = a.get('href')
        if href and 'tel:' in href:
            tel = href.split('tel:')[1]
        elif href and 'http' in href:
            website = href
        elif href and 'mailto:' in href:
            email = href.split('mailto:')[1]

    exhibitor_contact.append(tel)
    exhibitor_web.append(website)
    exhibitor_email.append(email)

    # Get exhibitor linkedin and twitter
    linkedin = ""
    twitter = ""

    social_links = bs.find_all('a', {'target': '_blank', 'rel': 'noopener noreferrer'})
    for a in social_links:
        href = a.get('href')
        if href and 'linkedin' in href:
            linkedin = href
        elif href and 'x.com' in href:
            twitter = href

    exhibitor_twitter.append(twitter)
    exhibitor_linkedin.append(linkedin)

    # Get booth number
    booth = ""
    booth_div = bs.find('div', class_='sc-901e7a18-2 eEIcVK')
    if booth_div:
        booth_span = booth_div.find('span', class_='sc-c3d23e77-0 epfMQJ')
        if booth_span:
            booth = booth_span.text
    
    exhibitor_booth.append(booth)

# Create a DataFrame
exhibitors_df = pd.DataFrame({
    'Exhibitor Name': exhibitor_names,
    'Event Info' : exhibitor_booth,
    'Description': exhibitor_info,
    'Website': exhibitor_web,
    'Linkedin' : exhibitor_linkedin,
    'Twitter' : exhibitor_twitter,
    'Email' : exhibitor_email,
    'Contact' : exhibitor_contact
})
解决方案

1. 安装依赖及配置驱动

首先安装Selenium库:

pip install selenium

同时下载对应浏览器的驱动(以Chrome为例),确保驱动版本与你的Chrome浏览器版本匹配,将驱动放在系统环境变量可访问的路径,或者在代码中指定路径。

2. 集成Selenium的完整代码

核心是用Selenium加载页面并滚动到底部触发动态加载,替换原有的requests.get获取首页内容的逻辑,其余抓取详情的代码可保留:

from selenium import webdriver
import time
from bs4 import BeautifulSoup
import pandas as pd
import requests

# 初始化Chrome浏览器(若驱动不在环境变量,需指定executable_path,如webdriver.Chrome(executable_path='./chromedriver.exe'))
driver = webdriver.Chrome()
driver.get('https://asiatechxsg.com/exhibitors/')

# 滚动页面加载所有动态内容
last_height = driver.execute_script("return document.body.scrollHeight")
while True:
    driver.execute_script("window.scrollTo(0, document.body.scrollHeight);")
    time.sleep(3)  # 等待加载,可根据网络情况调整时长
    new_height = driver.execute_script("return document.body.scrollHeight")
    if new_height == last_height:
        break
    last_height = new_height

# 获取加载完成后的页面源码并关闭浏览器
html = driver.page_source
driver.quit()

# 解析页面筛选参展商链接(过滤无关链接)
bs = BeautifulSoup(html, 'html.parser')
exhibitor_links = []
for link in bs.find_all('a'):
    if link.has_attr('href'):
        href = link.attrs['href']
        # 筛选参展商详情页特征链接,避免导航栏等无效链接
        if '/e/' in href:
            exhibitor_links.append(href)

# 以下保留原有详情抓取逻辑
exhibitor_names = []
exhibitor_info = []
exhibitor_web = []
exhibitor_linkedin = []
exhibitor_twitter = []
exhibitor_email = []
exhibitor_contact = []
exhibitor_booth = []

base_url = "https://attend.informatechevents.virtual.informatech.com/"
for link in exhibitor_links:
    url = base_url + link
    # 优先用requests获取,失败则用Selenium加载详情页
    try:
        bs = BeautifulSoup(requests.get(url).text, 'html.parser')
    except:
        driver_detail = webdriver.Chrome()
        driver_detail.get(url)
        time.sleep(2)
        bs = BeautifulSoup(driver_detail.page_source, 'html.parser')
        driver_detail.quit()
 
    name = bs.find('h1', class_= "sc-c418aba9-7 dbnofQ").text if bs.find('h1', class_= "sc-c418aba9-7 dbnofQ") else ""
    info_div = bs.find('div', class_='sc-dd6f9f7c-0 hIHCKf')

    if info_div:
        info_paras= info_div.find_all('p')
        info = '\n'.join([p.get_text() for p in info_paras])
        exhibitor_info.append(info)
    else:
        exhibitor_info.append("")
    exhibitor_names.append(name)

    tel = ""
    website = ""
    email = ""
    contact_links = bs.find_all('a', class_='sc-2d6f1d18-0 isnIyX')
    for a in contact_links:
        href = a.get('href')
        if href and 'tel:' in href:
            tel = href.split('tel:')[1]
        elif href and 'http' in href:
            website = href
        elif href and 'mailto:' in href:
            email = href.split('mailto:')[1]

    exhibitor_contact.append(tel)
    exhibitor_web.append(website)
    exhibitor_email.append(email)

    linkedin = ""
    twitter = ""
    social_links = bs.find_all('a', {'target': '_blank', 'rel': 'noopener noreferrer'})
    for a in social_links:
        href = a.get('href')
        if href and 'linkedin' in href:
            linkedin = href
        elif href and 'x.com' in href:
            twitter = href

    exhibitor_twitter.append(twitter)
    exhibitor_linkedin.append(linkedin)

    booth = ""
    booth_div = bs.find('div', class_='sc-901e7a18-2 eEIcVK')
    if booth_div:
        booth_span = booth_div.find('span', class_='sc-c3d23e77-0 epfMQJ')
        if booth_span:
            booth = booth_span.text
    
    exhibitor_booth.append(booth)

# 生成DataFrame并保存为CSV
exhibitors_df = pd.DataFrame({
    'Exhibitor Name': exhibitor_names,
    'Event Info' : exhibitor_booth,
    'Description': exhibitor_info,
    'Website': exhibitor_web,
    'Linkedin' : exhibitor_linkedin,
    'Twitter' : exhibitor_twitter,
    'Email' : exhibitor_email,
    'Contact' : exhibitor_contact
})
exhibitors_df.to_csv('exhibitors.csv', index=False, encoding='utf-8-sig')
print("数据已保存到exhibitors.csv")

关键说明

  • 滚动加载:通过循环滚动页面并等待,直到页面高度不再变化,确保所有动态加载的参展商链接都被加载。
  • 链接筛选:添加/e/特征过滤,避免抓取导航栏等无关链接,减少无效请求。
  • 详情页兼容:加入异常处理,若requests无法获取详情页内容,自动切换为Selenium加载。
  • 驱动路径:如果驱动未在环境变量中,需在初始化webdriver.Chrome()时指定executable_path参数。

内容的提问来源于stack exchange,提问作者user22279494

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.24 01:19:53