使用Python BeautifulSoup提取指定div全量数据的抗变更方案咨询
BscScan代币信息爬取优化实现
优化说明
- 核心逻辑优先定位
row mb-4父容器,所有字段仅从该容器内提取,和页面其他区域完全隔离 - 采用关键词匹配提取字段,不依赖固定的元素ID、嵌套层级,页面局部结构调整不会导致整体失效
- 修复原有代码语法错误、重复请求问题,增加异常捕获逻辑,单个字段提取失败不影响整体运行
完整代码
import requests import re from bs4 import BeautifulSoup # 全局请求配置 HEADERS = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:109.0) Gecko/20100101 Firefox/115.0' } BASE_URL = "https://bscscan.com/token/" TOKEN_ADDR = "0x4ce1a5cb12151423ea479cfd0c52ec5021d108d8" def extract_field(container, keyword): """从目标容器中按关键词提取对应值,抗结构变动""" try: ele = container.find(string=re.compile(keyword, re.IGNORECASE)) if ele: parent = ele.find_parent() return parent.get_text(strip=True).replace(keyword, "").strip() except Exception: return "提取失败" def get_token_info(token_addr): with requests.Session() as s: s.headers = HEADERS # 单次请求页面,复用会话 resp = s.get(f"{BASE_URL}{token_addr}", timeout=10) resp.raise_for_status() soup = BeautifulSoup(resp.text, 'html.parser') # 锁定目标容器,所有数据仅从该容器提取 target_container = soup.find('div', class_=['row', 'mb-4']) if not target_container: raise ValueError("未找到目标数据区块") # 基础字段提取 token_info = {} token_info['Token'] = extract_field(target_container, "Token Name:") or extract_field(target_container, "Token:") token_info['PRICE'] = extract_field(target_container, "Price:") token_info['Fully Diluted Market Cap'] = extract_field(target_container, "Fully Diluted Market Cap:") token_info['Total Supply'] = extract_field(target_container, "Total Supply:") token_info['Holders'] = extract_field(target_container, "Holders:") token_info['Contract'] = token_addr token_info['Decimals'] = extract_field(target_container, "Decimals:") token_info['Official Site'] = extract_field(target_container, "Official Site:") # 提取转账数 try: sid = re.search(r"var sid = '(.*?)'", resp.text).group(1) tx_resp = s.get(f'https://bscscan.com/token/generic-tokentxns2?m=normal&contractAddress={token_addr}&a=&sid={sid}&p=1', timeout=10) token_info['Transfers'] = re.search(r"var totaltxns = '(.*?)'", tx_resp.text).group(1) except Exception: token_info['Transfers'] = "提取失败" # 提取社交链接 social_links = [] for a_tag in target_container.find_all('a', href=True): href = a_tag['href'] if href.startswith('http') and 'bscscan.com' not in href: if href != token_info.get('Official Site', ''): social_links.append(href) token_info['Social Profiles'] = list(set(social_links)) return token_info if __name__ == "__main__": try: info = get_token_info(TOKEN_ADDR) # 按期望格式输出 print(f"Token: {info['Token']}") print(f"PRICE: {info['PRICE']}") print(f"Fully Diluted Market Cap: {info['Fully Diluted Market Cap']}") print() print(f"Total Supply: {info['Total Supply']}") print(f"Holders: {info['Holders']} addresses") print(f"Transfers: {info['Transfers']}") print(f"Contract: {info['Contract']}") print(f"Decimals: {info['Decimals']}") print(f"Official Site: {info['Official Site']}") print("Social Profiles:") for link in info['Social Profiles']: print(f" {link}") except Exception as e: print(f"运行出错: {str(e)}")
注意事项
- BscScan有反爬限制,请求频率不要过高,否则会触发验证码拦截
- 若页面字段名发生调整,只需修改
extract_field传入的关键词即可,无需修改整体逻辑
内容的提问来源于stack exchange,提问作者rbutrnz
相关产品推荐
相关产品推荐

