使用Beautiful Soup解析条目列表并保存至DataFrame遇问题求助
全球天主教会教区数据采集方案(BS4+Pandas)
问题背景
正在通过BS4和Pandas采集全球天主教会教区数据,编写的爬虫代码运行时出现异常,需要一套完整可行的采集实现方案。
原代码如下:
import requests from bs4 import BeautifulSoup import pandas as pd url = "http://www.catholic-hierarchy.org/" # Send a GET request to the website response = requests.get(url) #my approach to parse the HTML content of the page soup = BeautifulSoup(response.text, 'html.parser') # Find the relevant elements containing diocese information diocese_elements = soup.find_all("div", class_="diocesan") # Initialize empty lists to store data dioceses = [] addresses = [] # Extract now data from each diocese element for diocese_element in diocese_elements: # Example: Extracting diocese name diocese_name = diocese_element.find("a").text.strip() dioceses.append(diocese_name) # Example: Extracting address address = diocese_element.find("div", class_="address").text.strip() addresses.append(address) # to save the whole data we create a DataFrame using pandas data = {'Diocese': dioceses, 'Address': addresses} df = pd.DataFrame(data) # Display the DataFrame print(df)
问题分析
原代码存在几个核心问题:
- 目标网站首页无
diocesan类元素,该类元素实际存在于国家/地区的教区列表页,直接爬首页无法获取有效数据 - 未处理元素查找失败的情况(如找不到
<a>或address类div时,调用.text会抛出AttributeError) - 未处理请求失败的场景(如网络错误、HTTP状态码异常)
- 未实现多级页面遍历,无法覆盖全球所有教区数据
完整采集方案代码
import requests from bs4 import BeautifulSoup import pandas as pd from time import sleep # 全局配置 BASE_URL = "http://www.catholic-hierarchy.org/" HEADERS = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/114.0.0.0 Safari/537.36" } # 存储所有教区数据 all_diocese_data = [] def fetch_page(url): """封装页面请求,处理异常""" try: response = requests.get(url, headers=HEADERS) response.raise_for_status() # 触发HTTP错误异常 return BeautifulSoup(response.text, 'html.parser') except Exception as e: print(f"请求失败 {url}: {str(e)}") return None def parse_country_page(country_url): """解析国家页面,提取教区数据""" soup = fetch_page(country_url) if not soup: return # 查找所有教区元素 diocese_rows = soup.find_all("tr", class_="diocesan") for row in diocese_rows: diocese_info = {} # 提取教区名称 name_tag = row.find("td", class_="left").find("a") diocese_info["教区名称"] = name_tag.text.strip() if name_tag else "未知" # 提取教区链接(可选) diocese_info["教区链接"] = BASE_URL + name_tag["href"] if (name_tag and "href" in name_tag.attrs) else "" # 提取主教座堂 cathedral_tag = row.find("td", class_="left", string=True) diocese_info["主教座堂"] = cathedral_tag.text.strip() if cathedral_tag else "" # 提取地址(部分教区无公开地址,需处理) address_tag = row.find("td", class_="left", colspan="2") diocese_info["地址"] = address_tag.text.strip() if address_tag else "" # 提取国家(从页面标题或路径获取) country_name = soup.find("h1").text.strip().replace("Dioceses of ", "") diocese_info["国家/地区"] = country_name all_diocese_data.append(diocese_info) sleep(0.5) # 控制请求频率,避免被封禁 def parse_continent_page(continent_url): """解析大洲页面,获取所有国家链接""" soup = fetch_page(continent_url) if not soup: return # 查找所有国家链接 country_links = soup.find_all("a", href=True) for link in country_links: href = link["href"] # 筛选国家页面链接(链接格式为/country/xxx.html) if href.startswith("/country/") and href.endswith(".html"): country_full_url = BASE_URL + href.lstrip("/") print(f"正在爬取国家页面: {country_full_url}") parse_country_page(country_full_url) sleep(1) # 大洲到国家的请求间隔 def main(): """主函数:从首页开始遍历所有大洲""" home_soup = fetch_page(BASE_URL) if not home_soup: return # 查找所有大洲链接 continent_links = home_soup.find_all("a", href=True) for link in continent_links: href = link["href"] # 筛选大洲页面链接(链接格式为/continent/xxx.html) if href.startswith("/continent/") and href.endswith(".html"): continent_full_url = BASE_URL + href.lstrip("/") print(f"正在爬取大洲页面: {continent_full_url}") parse_continent_page(continent_full_url) sleep(1) # 将数据转换为DataFrame并保存 df = pd.DataFrame(all_diocese_data) # 保存为CSV文件 df.to_csv("全球天主教会教区数据.csv", index=False, encoding="utf-8-sig") # 保存为Excel文件(可选) # df.to_excel("全球天主教会教区数据.xlsx", index=False) print(f"数据采集完成,共获取{len(df)}条教区数据,已保存至文件") if __name__ == "__main__": main()
关键说明
- 请求封装:
fetch_page函数统一处理请求异常,避免单个请求失败导致程序崩溃 - 多级页面遍历:从首页→大洲页→国家页→教区数据,覆盖全球所有教区
- 异常处理:对所有元素查找操作添加空值判断,避免
AttributeError - 请求频率控制:添加
sleep控制请求间隔,防止触发网站反爬机制 - 数据存储:支持CSV/Excel格式保存,
utf-8-sig编码确保中文正常显示
内容的提问来源于stack exchange,提问作者zero
相关产品推荐
相关产品推荐

